diff --git a/.gitattributes b/.gitattributes index a6344aac8c09253b3b630fb776ae94478aa0275b..72a6c2d6492e1f601cf0f427047849edfc5d780a 100644 --- a/.gitattributes +++ b/.gitattributes @@ -33,3 +33,9 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text *.zip filter=lfs diff=lfs merge=lfs -text *.zst filter=lfs diff=lfs merge=lfs -text *tfevents* filter=lfs diff=lfs merge=lfs -text +wandb/run-20260809_035819-cvzjg5ej/run-cvzjg5ej.wandb filter=lfs diff=lfs merge=lfs -text +wandb/run-20260809_050050-59pftr14/run-59pftr14.wandb filter=lfs diff=lfs merge=lfs -text +wandb/run-20260809_055344-aqwnomdl/run-aqwnomdl.wandb filter=lfs diff=lfs merge=lfs -text +wandb/run-20260809_055726-m1dnjnh6/run-m1dnjnh6.wandb filter=lfs diff=lfs merge=lfs -text +wandb/run-20260809_061951-oe9tdw54/run-oe9tdw54.wandb filter=lfs diff=lfs merge=lfs -text +wandb/run-20260809_070213-uvqyddz0/run-uvqyddz0.wandb filter=lfs diff=lfs merge=lfs -text diff --git a/outio/mlp-linear-9L_run/checkpoint-750/scheduler.pt b/outio/mlp-linear-9L_run/checkpoint-750/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..04d5fe0a3a0a9c06aadc41faf74080bff4bcfc55 --- /dev/null +++ b/outio/mlp-linear-9L_run/checkpoint-750/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ce4f81803861286fbc66ef29ca3a3f867174d8f7eaf0b4091acb991c9eaf7690 +size 1064 diff --git a/outio/mlp-linear-9L_run/checkpoint-750/tokenizer.json b/outio/mlp-linear-9L_run/checkpoint-750/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..4e4e61f2579f8f18013caa72cd48f66b9ed1ad9f --- /dev/null +++ b/outio/mlp-linear-9L_run/checkpoint-750/tokenizer.json @@ -0,0 +1,19501 @@ +{ + "version": "1.0", + "truncation": null, + "padding": null, + "added_tokens": [ + { + "id": 0, + "content": "<|endoftext|>", + "single_word": false, + "lstrip": false, + "rstrip": false, + "normalized": true, + "special": true + } + ], + "normalizer": null, + "pre_tokenizer": { + "type": "ByteLevel", + "add_prefix_space": false, + "trim_offsets": true, + "use_regex": true + }, + "post_processor": { + "type": "ByteLevel", + "add_prefix_space": true, + "trim_offsets": false, + "use_regex": true + }, + "decoder": { + "type": "ByteLevel", + "add_prefix_space": true, + "trim_offsets": true, + "use_regex": true + }, + "model": { + "type": "BPE", + "dropout": null, + "unk_token": null, + "continuing_subword_prefix": "", + "end_of_word_suffix": "", + "fuse_unk": false, + "byte_fallback": false, + "ignore_merges": false, + "vocab": { + "<|endoftext|>": 0, + "!": 1, + "\"": 2, + "#": 3, + "$": 4, + "%": 5, + "&": 6, + "'": 7, + "(": 8, + ")": 9, + "*": 10, + "+": 11, + ",": 12, + "-": 13, + ".": 14, + "/": 15, + "0": 16, + "1": 17, + "2": 18, + "3": 19, + "4": 20, + "5": 21, + "6": 22, + "7": 23, + "8": 24, + "9": 25, + ":": 26, + ";": 27, + "<": 28, + "=": 29, + ">": 30, + "?": 31, + "@": 32, + "A": 33, + "B": 34, + "C": 35, + "D": 36, + "E": 37, + "F": 38, + "G": 39, + "H": 40, + "I": 41, + "J": 42, + "K": 43, + "L": 44, + "M": 45, + "N": 46, + "O": 47, + "P": 48, + "Q": 49, + "R": 50, + "S": 51, + "T": 52, + "U": 53, + "V": 54, + "W": 55, + "X": 56, + "Y": 57, + "Z": 58, + "[": 59, + "\\": 60, + "]": 61, + "^": 62, + "_": 63, + "`": 64, + "a": 65, + "b": 66, + "c": 67, + "d": 68, + "e": 69, + "f": 70, + "g": 71, + "h": 72, + "i": 73, + "j": 74, + "k": 75, + "l": 76, + "m": 77, + "n": 78, + "o": 79, + "p": 80, + "q": 81, + "r": 82, + "s": 83, + "t": 84, + "u": 85, + "v": 86, + "w": 87, + "x": 88, + "y": 89, + "z": 90, + "{": 91, + "|": 92, + "}": 93, + "~": 94, + "¡": 95, + "¢": 96, + "£": 97, + "¤": 98, + "¥": 99, + "¦": 100, + "§": 101, + "¨": 102, + "©": 103, + "ª": 104, + "«": 105, + "¬": 106, + "®": 107, + "¯": 108, + "°": 109, + "±": 110, + "²": 111, + "³": 112, + "´": 113, + "µ": 114, + "¶": 115, + "·": 116, + "¸": 117, + "¹": 118, + "º": 119, + "»": 120, + "¼": 121, + "½": 122, + "¾": 123, + "¿": 124, + "À": 125, + "Á": 126, + "Â": 127, + "Ã": 128, + "Ä": 129, + "Å": 130, + "Æ": 131, + "Ç": 132, + "È": 133, + "É": 134, + "Ê": 135, + "Ë": 136, + "Ì": 137, + "Í": 138, + "Î": 139, + "Ï": 140, + "Ð": 141, + "Ñ": 142, + "Ò": 143, + "Ó": 144, + "Ô": 145, + "Õ": 146, + "Ö": 147, + "×": 148, + "Ø": 149, + "Ù": 150, + "Ú": 151, + "Û": 152, + "Ü": 153, + "Ý": 154, + "Þ": 155, + "ß": 156, + "à": 157, + "á": 158, + "â": 159, + "ã": 160, + "ä": 161, + "å": 162, + "æ": 163, + "ç": 164, + "è": 165, + "é": 166, + "ê": 167, + "ë": 168, + "ì": 169, + "í": 170, + "î": 171, + "ï": 172, + "ð": 173, + "ñ": 174, + "ò": 175, + "ó": 176, + "ô": 177, + "õ": 178, + "ö": 179, + "÷": 180, + "ø": 181, + "ù": 182, + "ú": 183, + "û": 184, + "ü": 185, + "ý": 186, + "þ": 187, + "ÿ": 188, + "Ā": 189, + "ā": 190, + "Ă": 191, + "ă": 192, + "Ą": 193, + "ą": 194, + "Ć": 195, + "ć": 196, + "Ĉ": 197, + "ĉ": 198, + "Ċ": 199, + "ċ": 200, + "Č": 201, + "č": 202, + "Ď": 203, + "ď": 204, + "Đ": 205, + "đ": 206, + "Ē": 207, + "ē": 208, + "Ĕ": 209, + "ĕ": 210, + "Ė": 211, + "ė": 212, + "Ę": 213, + "ę": 214, + "Ě": 215, + "ě": 216, + "Ĝ": 217, + "ĝ": 218, + "Ğ": 219, + "ğ": 220, + "Ġ": 221, + "ġ": 222, + "Ģ": 223, + "ģ": 224, + "Ĥ": 225, + "ĥ": 226, + "Ħ": 227, + "ħ": 228, + "Ĩ": 229, + "ĩ": 230, + "Ī": 231, + "ī": 232, + "Ĭ": 233, + "ĭ": 234, + "Į": 235, + "į": 236, + "İ": 237, + "ı": 238, + "IJ": 239, + "ij": 240, + "Ĵ": 241, + "ĵ": 242, + "Ķ": 243, + "ķ": 244, + "ĸ": 245, + "Ĺ": 246, + "ĺ": 247, + "Ļ": 248, + "ļ": 249, + "Ľ": 250, + "ľ": 251, + "Ŀ": 252, + "ŀ": 253, + "Ł": 254, + "ł": 255, + "Ń": 256, + "he": 257, + "Ġt": 258, + "Ġa": 259, + "Ġs": 260, + "Ġw": 261, + "nd": 262, + "Ġthe": 263, + "ed": 264, + "Ġand": 265, + "Ġto": 266, + "Ġb": 267, + "in": 268, + "Ġh": 269, + "Ġwa": 270, + "re": 271, + "Ġf": 272, + "it": 273, + "ou": 274, + "Ġc": 275, + "Ġl": 276, + "Ġhe": 277, + "Ġd": 278, + "er": 279, + "Ġwas": 280, + "Ġm": 281, + "Ġp": 282, + "om": 283, + "ĠT": 284, + "Ġo": 285, + "ay": 286, + "ar": 287, + "ing": 288, + "is": 289, + "Ġg": 290, + "il": 291, + "id": 292, + "at": 293, + "en": 294, + "Ġn": 295, + "Ġsa": 296, + "Ġha": 297, + "ĠS": 298, + "im": 299, + "an": 300, + "ĠThe": 301, + "or": 302, + "on": 303, + "Ġit": 304, + "Ġth": 305, + "ll": 306, + "le": 307, + "ĠH": 308, + "Ġher": 309, + "et": 310, + "ot": 311, + "ir": 312, + "ĠShe": 313, + "ĠHe": 314, + "ver": 315, + "es": 316, + "Ġin": 317, + "ut": 318, + "ow": 319, + "ck": 320, + "Ġe": 321, + "Ġu": 322, + "ld": 323, + "ĠThey": 324, + "oo": 325, + "ig": 326, + "Ġsaid": 327, + "am": 328, + "ily": 329, + "Ġbe": 330, + "Ġy": 331, + "Ġr": 332, + "Ġst": 333, + "ce": 334, + "Ġshe": 335, + "Ġ\"": 336, + "pp": 337, + "ke": 338, + "ith": 339, + "On": 340, + "ĠI": 341, + "Ġwith": 342, + "ve": 343, + "Lily": 344, + "Ġon": 345, + "Ġof": 346, + "Ġso": 347, + "Ġhis": 348, + "ked": 349, + "ri": 350, + "nt": 351, + "very": 352, + "Ġpl": 353, + "Ġday": 354, + "ad": 355, + "Ġyou": 356, + "Ġthat": 357, + "Ġup": 358, + "Ġhad": 359, + "st": 360, + "Ġplay": 361, + "Ġthey": 362, + "ĠLily": 363, + "Ġwe": 364, + "Ġmom": 365, + "my": 366, + "Ġfor": 367, + "el": 368, + "ould": 369, + "un": 370, + "ĠB": 371, + "'s": 372, + "itt": 373, + "ent": 374, + "Ġhapp": 375, + "The": 376, + "ch": 377, + "Ġli": 378, + "out": 379, + "Ġwant": 380, + "Ġsh": 381, + "her": 382, + "ly": 383, + "ime": 384, + "ittle": 385, + "ound": 386, + "Ġvery": 387, + "Ġtime": 388, + "ome": 389, + "Ġlittle": 390, + "Ġthere": 391, + "se": 392, + "Ġdo": 393, + "Ġwh": 394, + "all": 395, + "Ġk": 396, + "end": 397, + "al": 398, + "ht": 399, + "Ġne": 400, + "Ġre": 401, + "Ġnot": 402, + "Ġhappy": 403, + "ĠĊ": 404, + "Ġbig": 405, + "ĠM": 406, + "Ġbut": 407, + "Ġsm": 408, + "ack": 409, + "Ġsaw": 410, + "ĠIt": 411, + "Ġas": 412, + "Ġan": 413, + "ra": 414, + "riend": 415, + "Ġfriend": 416, + "ide": 417, + "One": 418, + "ry": 419, + "'t": 420, + "ved": 421, + "Ġis": 422, + "Once": 423, + "ake": 424, + ".\"": 425, + "Ġwere": 426, + "ter": 427, + "ug": 428, + "Ġloo": 429, + "Ġlo": 430, + "ore": 431, + "ec": 432, + "ĠTim": 433, + "Ġhim": 434, + "Ġbo": 435, + "!\"": 436, + "Ġtoo": 437, + "Ġgo": 438, + "Ġupon": 439, + "irl": 440, + "Ġj": 441, + "Ġwanted": 442, + "Ġgirl": 443, + "Ġse": 444, + "Ġout": 445, + "ard": 446, + "way": 447, + "Ġsp": 448, + "ill": 449, + "ind": 450, + "Ġthem": 451, + "Ġcould": 452, + "fu": 453, + "hen": 454, + "Ġat": 455, + "ur": 456, + "Ġdid": 457, + "Ġsmil": 458, + "Ġtheir": 459, + "Ġare": 460, + "Ġex": 461, + "ĠA": 462, + "ain": 463, + "Ġwent": 464, + "art": 465, + "hed": 466, + "rom": 467, + "ic": 468, + "round": 469, + "Ġhave": 470, + "Ġnam": 471, + "lp": 472, + "Ġall": 473, + "ĠJ": 474, + "ful": 475, + "Ġkn": 476, + "hing": 477, + "ood": 478, + "Ġhelp": 479, + "ight": 480, + "Ġfriends": 481, + "one": 482, + "ark": 483, + "Ġback": 484, + "um": 485, + "Ġcan": 486, + "Ġnamed": 487, + "Ġcl": 488, + "?\"": 489, + "Ġfun": 490, + "are": 491, + "ĠBen": 492, + "Ġloved": 493, + "Ġal": 494, + "elt": 495, + "ĠTimmy": 496, + "ĠOne": 497, + "op": 498, + "side": 499, + "Ġle": 500, + "Ġno": 501, + "Ġsc": 502, + "ĠTom": 503, + "Ġfelt": 504, + "oug": 505, + "Ġsmiled": 506, + "ick": 507, + "Ġasked": 508, + "You": 509, + "Ġtoy": 510, + "Ġman": 511, + "Ġaround": 512, + "ame": 513, + "Ġfe": 514, + "Ġsay": 515, + "Ġboy": 516, + "Ġsome": 517, + "Ġlooked": 518, + "ure": 519, + "omet": 520, + "Ġbr": 521, + "Ġwould": 522, + "Ġme": 523, + "Ġbir": 524, + "Ġlike": 525, + "get": 526, + "Ġstart": 527, + "Ġro": 528, + "as": 529, + "Ġsee": 530, + "ĠW": 531, + "ice": 532, + "ong": 533, + "Ġbird": 534, + "Ġsomet": 535, + "dd": 536, + "Ġwor": 537, + "ade": 538, + "ie": 539, + "king": 540, + "Ġag": 541, + "own": 542, + "Ġtre": 543, + "Ġfa": 544, + "Ġaway": 545, + "Ġwhat": 546, + "ings": 547, + "Ġstarted": 548, + "gether": 549, + "Ġran": 550, + "â": 551, + "âĤ": 552, + "âĤ¬": 553, + "ared": 554, + "Ġmake": 555, + "ĠBut": 556, + "ited": 557, + "if": 558, + "oud": 559, + "Ġmade": 560, + "Ġtogether": 561, + "Ġsomething": 562, + "Ġexc": 563, + "ag": 564, + "Ġco": 565, + "Ġpark": 566, + "Ġnew": 567, + "Ġsad": 568, + "Ġput": 569, + ",\"": 570, + "Ġfrom": 571, + "ble": 572, + "ther": 573, + "Ġpr": 574, + "Ġmu": 575, + "Ġcar": 576, + "Ġhome": 577, + "ĠYou": 578, + "Ġthen": 579, + "Ġwhen": 580, + "Ġfound": 581, + "ell": 582, + "Ġother": 583, + "Ġagain": 584, + "Ġch": 585, + "Ġdec": 586, + "Ġwho": 587, + "Ġla": 588, + "ried": 589, + "ss": 590, + "Ġgood": 591, + "Ġhug": 592, + "ĠL": 593, + "pped": 594, + "Ġwal": 595, + "ep": 596, + "ally": 597, + "Ġsays": 598, + "Ġfl": 599, + "est": 600, + "ach": 601, + "ĠE": 602, + "Ġexcited": 603, + "pl": 604, + "qu": 605, + "ook": 606, + "Ġget": 607, + "ought": 608, + "Ġplaying": 609, + "Ġgot": 610, + "Ġsw": 611, + "ous": 612, + "hat": 613, + "ny": 614, + "ided": 615, + "uck": 616, + "Ġthings": 617, + "Ġevery": 618, + "Ġdecided": 619, + "Ġcame": 620, + "Ġbec": 621, + "ave": 622, + "ro": 623, + "ax": 624, + "Ġliked": 625, + "Ġdown": 626, + "Ġdog": 627, + "Ġscared": 628, + "Ġv": 629, + "udd": 630, + "ust": 631, + "Ġone": 632, + "Ġfind": 633, + "Ġbl": 634, + "Ġthan": 635, + "ĠD": 636, + "ouse": 637, + "ways": 638, + "Ġkne": 639, + "Ġdidn": 640, + "ap": 641, + "Ġmommy": 642, + "Ġcare": 643, + "Ġalways": 644, + "Ġab": 645, + "ist": 646, + "Ġdad": 647, + "Ġfeel": 648, + "ara": 649, + "bb": 650, + "arn": 651, + "fe": 652, + "Ġyour": 653, + "Ġoutside": 654, + "ue": 655, + "Ġgra": 656, + "ant": 657, + "Ġke": 658, + "ĠMom": 659, + "Ġtook": 660, + "Ġlot": 661, + "nn": 662, + "Ġbu": 663, + "Ġabout": 664, + "ess": 665, + "ĠF": 666, + "nder": 667, + "Ġtree": 668, + "eci": 669, + "Ġlook": 670, + "Ġpo": 671, + "Ġmy": 672, + "our": 673, + "Ġtoys": 674, + "ite": 675, + "ched": 676, + "Ġknew": 677, + "Ġthought": 678, + "ened": 679, + "Ġlearn": 680, + "Ġint": 681, + "Ġold": 682, + "Ġmore": 683, + "nna": 684, + "ise": 685, + "ged": 686, + "Ġta": 687, + "udden": 688, + "ecial": 689, + "Ġspecial": 690, + "ĠMax": 691, + "au": 692, + "Ġwill": 693, + "ers": 694, + "They": 695, + "ret": 696, + "Ġpe": 697, + "Ġho": 698, + "ĠSam": 699, + "Ġtake": 700, + "Ġball": 701, + "Ġknow": 702, + "Ġlaug": 703, + "fter": 704, + "uddenly": 705, + "Ġcat": 706, + "Ġhow": 707, + "ive": 708, + "Ġtr": 709, + "Ġmuch": 710, + "Ġany": 711, + "Ġpu": 712, + "ma": 713, + "Ġsl": 714, + "Ġsor": 715, + "Ġmo": 716, + "ven": 717, + "ish": 718, + "Ġshow": 719, + "Ġcouldn": 720, + "ause": 721, + "ink": 722, + "But": 723, + "Ġhouse": 724, + "Ġinto": 725, + "ump": 726, + "Ġover": 727, + "Ġtried": 728, + "and": 729, + "Ġeat": 730, + "Ġsk": 731, + "Ġsun": 732, + "Ġtw": 733, + "Ġclo": 734, + "ia": 735, + "Ġrun": 736, + "Ġhand": 737, + "ĠEvery": 738, + "Ġen": 739, + "Ġinside": 740, + "dy": 741, + "Ġif": 742, + "Ġtold": 743, + "Ġnever": 744, + "ion": 745, + "Ġqu": 746, + "Ġbecause": 747, + "by": 748, + "Ġproud": 749, + "Ġgave": 750, + "Ġsorry": 751, + "Ġthis": 752, + "Ġop": 753, + "Ġplayed": 754, + "ate": 755, + "Ġexpl": 756, + "Ġheard": 757, + "ty": 758, + "Ġor": 759, + "Ġwater": 760, + "ank": 761, + "ge": 762, + "sed": 763, + "Ġpick": 764, + "other": 765, + "Ġroom": 766, + "etter": 767, + "Ġjust": 768, + "ace": 769, + "Ġhugged": 770, + "ak": 771, + "Ġgre": 772, + "here": 773, + "Ġoff": 774, + "ĠSara": 775, + "Ġpret": 776, + "ile": 777, + "Ġeach": 778, + "Ġcom": 779, + "Ġlong": 780, + "Ġbox": 781, + "ort": 782, + "Ġstr": 783, + "iz": 784, + "Ġunt": 785, + "Ġwat": 786, + "oth": 787, + "Ġneed": 788, + "Ġjo": 789, + "ĠWe": 790, + "Tom": 791, + "Ġsmall": 792, + "ine": 793, + "Ġbear": 794, + "Mom": 795, + "Ġuntil": 796, + "Ġnice": 797, + "Ġtry": 798, + "ving": 799, + "uc": 800, + "sel": 801, + "ough": 802, + "Ġlearned": 803, + "Ġkind": 804, + "ĠAnna": 805, + "ild": 806, + "Ġfo": 807, + "Ġmany": 808, + "'m": 809, + "ĠJack": 810, + "Ġbetter": 811, + "Ġim": 812, + "gry": 813, + "imal": 814, + "Ġanimal": 815, + "urt": 816, + "ft": 817, + "Ġend": 818, + "Ġsn": 819, + "aut": 820, + "Ġte": 821, + "Ġcle": 822, + "ĠJo": 823, + "vent": 824, + "urp": 825, + "Ġgr": 826, + "Ġbeaut": 827, + "Ġjump": 828, + "mb": 829, + "Ġad": 830, + "ream": 831, + "pt": 832, + "ĠSo": 833, + "ĠHer": 834, + "Ġflow": 835, + "ies": 836, + "Ġche": 837, + "Ġbra": 838, + "Ġthanked": 839, + "Ġeven": 840, + "Ġbest": 841, + "Ġcall": 842, + "ady": 843, + "He": 844, + "Ġlots": 845, + "Ġlaughed": 846, + "self": 847, + "Ġra": 848, + "Ġway": 849, + "ars": 850, + "urn": 851, + "Ġby": 852, + "Ġfast": 853, + "lly": 854, + "Ġfam": 855, + "ĠC": 856, + "ves": 857, + "arden": 858, + "Ġgarden": 859, + "Ġbeauti": 860, + "wn": 861, + "Th": 862, + "Ġbeautiful": 863, + "Ġloud": 864, + "lew": 865, + "Ġsky": 866, + "Ġdon": 867, + "hn": 868, + "ered": 869, + "iny": 870, + "Ġcareful": 871, + "Ġlove": 872, + "Ġfi": 873, + "ĠThen": 874, + "Åĵ": 875, + "ase": 876, + "ect": 877, + "Ġsafe": 878, + "ĠAnd": 879, + "Ġunder": 880, + "Ġcome": 881, + "ĠFrom": 882, + "Yes": 883, + "ĠMia": 884, + "It": 885, + "me": 886, + "Ġhard": 887, + "Ġcu": 888, + "Ġwo": 889, + "Ġlist": 890, + "Ġstay": 891, + "ane": 892, + "sh": 893, + "ople": 894, + "Ġgl": 895, + "ning": 896, + "Ġstill": 897, + "ool": 898, + "Ġhurt": 899, + "ree": 900, + "ĠHis": 901, + "Ġimp": 902, + "Ġfamily": 903, + "Ġâ": 904, + "Ġboth": 905, + "rm": 906, + "igh": 907, + "Ġlived": 908, + "hes": 909, + "When": 910, + "Ġpeople": 911, + "Ġanimals": 912, + "Ġcol": 913, + "Ġbrave": 914, + "Ġwalked": 915, + "ob": 916, + "Tim": 917, + "ct": 918, + "Ġlet": 919, + "urpr": 920, + "ĠWhen": 921, + "Ġtwo": 922, + "Ġsurpr": 923, + "Ġshould": 924, + "ished": 925, + "Ġbad": 926, + "ress": 927, + "Ġkept": 928, + "Ġfore": 929, + "Ġflew": 930, + "Ġfin": 931, + "Ġstor": 932, + "Ġfly": 933, + "ast": 934, + "ised": 935, + "ip": 936, + "Ġits": 937, + "led": 938, + "ock": 939, + "ucy": 940, + "fore": 941, + "Ġgoing": 942, + "Ġclean": 943, + "Ġdan": 944, + "Ġpic": 945, + "Ġsoon": 946, + "Ġcalled": 947, + "Ġshare": 948, + "kay": 949, + "Ġangry": 950, + "Ġrock": 951, + "Ġcon": 952, + "Ġpretty": 953, + "No": 954, + "Ġide": 955, + "ied": 956, + "illy": 957, + "Ġground": 958, + "xt": 959, + "Ġred": 960, + "Ġexplore": 961, + "Ġcry": 962, + "Ġadvent": 963, + "Ġsto": 964, + "so": 965, + "Ġreal": 966, + "Let": 967, + "Ġwind": 968, + "Ġshiny": 969, + "be": 970, + "Ġbook": 971, + "Ġalso": 972, + "Ġdoll": 973, + "Ġidea": 974, + "Ġbefore": 975, + "Ġopened": 976, + "dded": 977, + "Ġwhile": 978, + "ummy": 979, + "Ġkeep": 980, + "Ġey": 981, + "Ġnow": 982, + "Ġdoor": 983, + "Ġfeeling": 984, + "âĤ¬â": 985, + "oon": 986, + "oy": 987, + "Ġwalking": 988, + "Ġnoise": 989, + "Ġfr": 990, + "les": 991, + "age": 992, + "ious": 993, + "Ġcolor": 994, + "Ġturn": 995, + "thing": 996, + "ff": 997, + "uch": 998, + "th": 999, + "Ġbed": 1000, + "ary": 1001, + "Ġdra": 1002, + "Ġpicked": 1003, + "imb": 1004, + "eet": 1005, + "Ġclimb": 1006, + "Ġdel": 1007, + "What": 1008, + "Ġbeing": 1009, + "Ġfood": 1010, + "Ġun": 1011, + "Ġfar": 1012, + "ture": 1013, + "joy": 1014, + "Ġadventure": 1015, + "ac": 1016, + "Ġsmile": 1017, + "Ġdif": 1018, + "Ħ¢": 1019, + "âĤ¬âĦ¢": 1020, + "memb": 1021, + "Ġwr": 1022, + "Ġthr": 1023, + "ught": 1024, + "Ġlooking": 1025, + "Ġnext": 1026, + "iced": 1027, + "ĠLucy": 1028, + "Ġnodded": 1029, + "Ġquick": 1030, + "ĠP": 1031, + "Ġdis": 1032, + "Ġrepl": 1033, + "ĠDad": 1034, + "Ġwait": 1035, + "ized": 1036, + "Ġforest": 1037, + "Ġclos": 1038, + "ĠSuddenly": 1039, + "Ġtra": 1040, + "That": 1041, + "Ġeyes": 1042, + "ger": 1043, + "bbit": 1044, + "ted": 1045, + "Ġown": 1046, + "Ġrain": 1047, + "Ġimport": 1048, + "Ġgreat": 1049, + "Ġrememb": 1050, + "Ġpicture": 1051, + "Thank": 1052, + "Suddenly": 1053, + "Ġstopped": 1054, + "Ġenjoy": 1055, + "Ġvo": 1056, + "Ben": 1057, + "Ġgive": 1058, + "Ġimportant": 1059, + "Ġwork": 1060, + "Ġnear": 1061, + "gan": 1062, + "pot": 1063, + "Ġever": 1064, + "Ġapp": 1065, + "Ġafter": 1066, + "Ġquickly": 1067, + "Ġlisten": 1068, + "Ġbre": 1069, + "ting": 1070, + "bbed": 1071, + "Ġma": 1072, + "Ġfish": 1073, + "Ġreplied": 1074, + "Ġrabbit": 1075, + "Ġhands": 1076, + "ĠG": 1077, + "Ġnoticed": 1078, + "Ġbro": 1079, + "Ġslide": 1080, + "Ġthink": 1081, + "Ġwalk": 1082, + "Ġtruck": 1083, + "Ġac": 1084, + "kes": 1085, + "Ġstrong": 1086, + "She": 1087, + "Ġshowed": 1088, + "Ġde": 1089, + "Ġeveryone": 1090, + "Ġwonder": 1091, + "fere": 1092, + "bye": 1093, + "Ġdiffere": 1094, + "irst": 1095, + "So": 1096, + "Ġsure": 1097, + "Ġhas": 1098, + "Ġright": 1099, + "Ġbeen": 1100, + "Ġbecame": 1101, + "Ġsound": 1102, + "maz": 1103, + "Ġtow": 1104, + "Ġru": 1105, + "Ġtal": 1106, + "Ġamaz": 1107, + "Ġhead": 1108, + "Ġshout": 1109, + "Ġbright": 1110, + "Ġye": 1111, + "Ġwatched": 1112, + "ĠR": 1113, + "Ġmor": 1114, + "Ġchild": 1115, + "able": 1116, + "Ġmean": 1117, + "Ġwhere": 1118, + "llow": 1119, + "Ġhigh": 1120, + "ĠSue": 1121, + "Ġface": 1122, + "Ġcook": 1123, + "day": 1124, + "aybe": 1125, + "Ġwatch": 1126, + "Ġblue": 1127, + "aught": 1128, + "Ġdifferent": 1129, + "Ġstore": 1130, + "ĠN": 1131, + "Ġgoodbye": 1132, + "Ġdress": 1133, + "ull": 1134, + "ĠBob": 1135, + "ng": 1136, + "ange": 1137, + "Ġsqu": 1138, + "Ġokay": 1139, + "isy": 1140, + "lease": 1141, + "ĠMommy": 1142, + "Ġvoice": 1143, + "Jo": 1144, + "ath": 1145, + "Ġnight": 1146, + "ĠSpot": 1147, + "Ġus": 1148, + "Ġboat": 1149, + "Ġflowers": 1150, + "Ġplace": 1151, + "Ġfollow": 1152, + "Ġar": 1153, + "Ġuse": 1154, + "Ġcloser": 1155, + "unny": 1156, + "leep": 1157, + "ired": 1158, + "Ġfav": 1159, + "Ġyell": 1160, + "Ġgrabbed": 1161, + "Ġcuri": 1162, + "Ġwarm": 1163, + "Ġcr": 1164, + "Ġforg": 1165, + "ĠSarah": 1166, + "ĠJohn": 1167, + "Ġmag": 1168, + "Ġstick": 1169, + "We": 1170, + "Ġjumped": 1171, + "Ġcake": 1172, + "more": 1173, + "Ġtell": 1174, + "Ġanymore": 1175, + "After": 1176, + "Ġbutter": 1177, + "ndma": 1178, + "Ġthree": 1179, + "Ġask": 1180, + "co": 1181, + "Ġour": 1182, + "lie": 1183, + "Ġcurious": 1184, + "ount": 1185, + "orn": 1186, + "Ġcont": 1187, + "Ġfell": 1188, + "ached": 1189, + "ĠTh": 1190, + "Ġbirds": 1191, + "ass": 1192, + "Her": 1193, + "iss": 1194, + "Ġhelped": 1195, + "ĠJane": 1196, + "Ġpull": 1197, + "Ġfirst": 1198, + "itc": 1199, + "Ġblock": 1200, + "Ġhop": 1201, + "Ġbit": 1202, + "Ġdr": 1203, + "Ġrealized": 1204, + "Ġkid": 1205, + "Look": 1206, + "ila": 1207, + "Ġmon": 1208, + "Ġbrother": 1209, + "Anna": 1210, + "Sara": 1211, + "Ġz": 1212, + "ĠEveryone": 1213, + "Ġate": 1214, + "Ġdoes": 1215, + "imes": 1216, + "Ġhappened": 1217, + "Ġstop": 1218, + "zy": 1219, + "Ġyummy": 1220, + "Ġfavor": 1221, + "ppy": 1222, + "Ġkitc": 1223, + "Ġkitchen": 1224, + "Ġsweet": 1225, + "us": 1226, + "Ġper": 1227, + "Ġreally": 1228, + "aisy": 1229, + "Ġgrass": 1230, + "Ġfavorite": 1231, + "Ġbegan": 1232, + "Ġrest": 1233, + "Ġready": 1234, + "Ġlea": 1235, + "Ġreached": 1236, + "Ġunderst": 1237, + "air": 1238, + "Ġste": 1239, + "Ġbunny": 1240, + "ĠAs": 1241, + "Ġstory": 1242, + "'re": 1243, + "Ġpain": 1244, + "Ġsing": 1245, + "ster": 1246, + "Ġhere": 1247, + "Ġsand": 1248, + "Ġonly": 1249, + "Ġflo": 1250, + "Ġam": 1251, + "Ġglad": 1252, + "Ġtri": 1253, + "Ġbeh": 1254, + "Ġworld": 1255, + "Ġopen": 1256, + "Ġcre": 1257, + "fully": 1258, + "Ġprin": 1259, + "where": 1260, + "arent": 1261, + "Ġflower": 1262, + "Ġthrough": 1263, + "Ġba": 1264, + "Ġfire": 1265, + "Ġdone": 1266, + "Ġhaving": 1267, + "Ġthing": 1268, + "Ġdelic": 1269, + "Ġhimself": 1270, + "Ġtired": 1271, + "Ġparent": 1272, + "Ġsoft": 1273, + "Ġfro": 1274, + "Timmy": 1275, + "Ġtast": 1276, + "Ġbutterf": 1277, + "ĠLet": 1278, + "Ġcut": 1279, + "Ġpart": 1280, + "Ġwhy": 1281, + "ken": 1282, + "Ġmess": 1283, + "Can": 1284, + "Ġworry": 1285, + "Mommy": 1286, + "iver": 1287, + "Ġdin": 1288, + "uff": 1289, + "ater": 1290, + "Ġmagic": 1291, + "Ġwaved": 1292, + "Ġshouted": 1293, + "Ġpond": 1294, + "Ġkids": 1295, + "Ġhat": 1296, + "Ġduck": 1297, + "Ġsees": 1298, + "olly": 1299, + "illed": 1300, + "Ġgame": 1301, + "ient": 1302, + "Ġmaking": 1303, + "ather": 1304, + "John": 1305, + "As": 1306, + "akes": 1307, + "Ġcatch": 1308, + "Ġseen": 1309, + "Ġcool": 1310, + "ation": 1311, + "Ġcoming": 1312, + "Ġless": 1313, + "Ġdark": 1314, + "Ġused": 1315, + "eddy": 1316, + "Ġfix": 1317, + "ĠJoe": 1318, + "Ġthank": 1319, + "mer": 1320, + "Ġtop": 1321, + "Ġlady": 1322, + "Ġhair": 1323, + "aring": 1324, + "Ġprom": 1325, + "Ġhopped": 1326, + "ĠCan": 1327, + "\".": 1328, + "ign": 1329, + "Ġsurprise": 1330, + "Ġdraw": 1331, + "Ġmum": 1332, + "Ġmouse": 1333, + "Ġfunny": 1334, + "rel": 1335, + "Ġcra": 1336, + "Ġ-": 1337, + "aper": 1338, + "Ġfull": 1339, + "Ġtouch": 1340, + "Ġlight": 1341, + "Ġspot": 1342, + "ren": 1343, + "Ġdro": 1344, + "ĠBenny": 1345, + "Ġparents": 1346, + "Ġworked": 1347, + "ins": 1348, + "oney": 1349, + "ĠDo": 1350, + "Ġsurprised": 1351, + "Ġcarefully": 1352, + "Ġtrees": 1353, + "Ġfrog": 1354, + "Ġswing": 1355, + "Ġdoing": 1356, + "Don": 1357, + "Ġsat": 1358, + "inally": 1359, + "ĠThat": 1360, + "Ġice": 1361, + "ĠIn": 1362, + "Ġheld": 1363, + "Wow": 1364, + "Ġrunning": 1365, + "Ġpretend": 1366, + "Ġswim": 1367, + "Ġset": 1368, + "Ġread": 1369, + "Ġdelicious": 1370, + "Ġneeded": 1371, + "Ġtight": 1372, + "Ġslow": 1373, + "Ġremembered": 1374, + "Ġlost": 1375, + "Ġcold": 1376, + "Ġsmell": 1377, + "ards": 1378, + "Ġwood": 1379, + "Ġhappily": 1380, + "hy": 1381, + "ely": 1382, + "Ġlooks": 1383, + "Ġbehind": 1384, + "Ġherself": 1385, + "Ġcried": 1386, + "Ġenjoyed": 1387, + "Ġname": 1388, + "ask": 1389, + "Ġyears": 1390, + "Ġbuy": 1391, + "Ġhole": 1392, + "Ġblocks": 1393, + "Ġdri": 1394, + "Ġsleep": 1395, + "Ġgi": 1396, + "Ġyellow": 1397, + "cess": 1398, + "Ġbutterfly": 1399, + "ike": 1400, + "ued": 1401, + "Ġwished": 1402, + "Ġperf": 1403, + "Ġanother": 1404, + "ĠDaisy": 1405, + "Ġgreen": 1406, + "Ġmove": 1407, + "Ġtall": 1408, + "Ġair": 1409, + "Ġbag": 1410, + "Ġfloor": 1411, + "Ġcars": 1412, + "Ġlesson": 1413, + "ul": 1414, + "Ġbow": 1415, + "Ġarri": 1416, + "Ġwindow": 1417, + "Ġsho": 1418, + "Ġchildren": 1419, + "ner": 1420, + "Ġfinished": 1421, + "Ġhill": 1422, + "ĠAfter": 1423, + "andy": 1424, + "sp": 1425, + "ens": 1426, + "Ġpaper": 1427, + "Ġlikes": 1428, + "From": 1429, + "sy": 1430, + "Ġhold": 1431, + "Sam": 1432, + "Ġwrong": 1433, + "reed": 1434, + "Ġcontin": 1435, + "Ġunderstand": 1436, + "atter": 1437, + "uit": 1438, + "rew": 1439, + "Ġgent": 1440, + "Ġanything": 1441, + "Ġclose": 1442, + "Ġwall": 1443, + "Ġel": 1444, + "asure": 1445, + "Ġleft": 1446, + "Ġable": 1447, + "Ġarrived": 1448, + "Ġhun": 1449, + "ross": 1450, + "av": 1451, + "Ġhear": 1452, + "Ġfriendly": 1453, + "Ġforgot": 1454, + "Ġgone": 1455, + "ama": 1456, + "Ġcreat": 1457, + "Ġwet": 1458, + "Ġlion": 1459, + "Hell": 1460, + "Hello": 1461, + "Ġdream": 1462, + "Ġhot": 1463, + "bo": 1464, + "ber": 1465, + "ĠSally": 1466, + "Ġfilled": 1467, + "Ġdir": 1468, + "ield": 1469, + "Ġcookies": 1470, + "Ġlaugh": 1471, + "Ġbroken": 1472, + "Ġdinner": 1473, + "lf": 1474, + "Ġelse": 1475, + "Ġhid": 1476, + "ced": 1477, + "Ġpink": 1478, + "Ġfollowed": 1479, + "Ġtable": 1480, + "Ġmar": 1481, + "Ġcolors": 1482, + "oup": 1483, + "Ġmoment": 1484, + "Ġcontinued": 1485, + "Ġwonderful": 1486, + "ĠEm": 1487, + "ey": 1488, + "ined": 1489, + "irrel": 1490, + "rot": 1491, + "Ġfinally": 1492, + "Jack": 1493, + "Ġarm": 1494, + "Ġmight": 1495, + "ĠNow": 1496, + "Ġcast": 1497, + "Ġyes": 1498, + "Ġbuild": 1499, + "Ġenough": 1500, + "ĠBilly": 1501, + "Ġsomeone": 1502, + "Ġperfect": 1503, + "Ġfair": 1504, + "Ġstayed": 1505, + "app": 1506, + "Ġmorning": 1507, + "ĠThere": 1508, + "Ġpuppy": 1509, + "ol": 1510, + "Ġdaddy": 1511, + "Ġtrying": 1512, + "'ll": 1513, + "Ġju": 1514, + "Ġothers": 1515, + "Ġbooks": 1516, + "Ġagreed": 1517, + "Ġmus": 1518, + "Ġsnow": 1519, + "Ġsquirrel": 1520, + "Ġcream": 1521, + "room": 1522, + "Ġgetting": 1523, + "Ġgrandma": 1524, + "ĠLila": 1525, + "Ġleaves": 1526, + "Ġbaby": 1527, + "Ġcount": 1528, + "icy": 1529, + "Ġfew": 1530, + "At": 1531, + "Ġfall": 1532, + "Then": 1533, + "gon": 1534, + "Ġbreak": 1535, + "Ġclimbed": 1536, + "ries": 1537, + "Ġbeach": 1538, + "ash": 1539, + "Ġbug": 1540, + "Ġplease": 1541, + "Ġmother": 1542, + "Ġbrought": 1543, + "Ġtrain": 1544, + "ance": 1545, + "ĠThis": 1546, + "Ġdeep": 1547, + "Ġvis": 1548, + "ated": 1549, + "shed": 1550, + "Ġmet": 1551, + "Ġwasn": 1552, + "Ġsometimes": 1553, + "Ġeverything": 1554, + "ĠTommy": 1555, + "ich": 1556, + "Ġmusic": 1557, + "key": 1558, + "ling": 1559, + "This": 1560, + "os": 1561, + "Ġturned": 1562, + "lc": 1563, + "Ġride": 1564, + "oom": 1565, + "Ġsong": 1566, + "oring": 1567, + "Ġspark": 1568, + "Ġfight": 1569, + "Ġhungry": 1570, + "ons": 1571, + "Ġpictures": 1572, + "ds": 1573, + "Ġmakes": 1574, + "Ġsear": 1575, + "ucky": 1576, + "Ġlistened": 1577, + "Ġjoy": 1578, + "Ġpol": 1579, + "ĠWhat": 1580, + "His": 1581, + "ee": 1582, + "Ġclot": 1583, + "Ġsail": 1584, + "apped": 1585, + "bble": 1586, + "Ġcomp": 1587, + "Ġplan": 1588, + "Ġwants": 1589, + "Ġclothes": 1590, + "Ġpoin": 1591, + "ĠK": 1592, + "zz": 1593, + "Ġballoon": 1594, + "Ġstrange": 1595, + "man": 1596, + "enny": 1597, + "Ġpi": 1598, + "Ġgrow": 1599, + "Ġflying": 1600, + "row": 1601, + "Ġscary": 1602, + "Ġtreat": 1603, + "Ġwaited": 1604, + "Ġamazed": 1605, + "Ġteddy": 1606, + "Ġcastle": 1607, + "ister": 1608, + "Ġele": 1609, + "Ġamazing": 1610, + "med": 1611, + "Ġspl": 1612, + "Oh": 1613, + "Ġleave": 1614, + "Ġowner": 1615, + "Ġpromised": 1616, + "Ġdanger": 1617, + "ever": 1618, + "OK": 1619, + "Ġsuch": 1620, + "Ġwhite": 1621, + "Ġpat": 1622, + "Ġtalk": 1623, + "uffy": 1624, + "Ġfur": 1625, + "Ġpie": 1626, + "ope": 1627, + "red": 1628, + "Ġpulled": 1629, + "Ġpre": 1630, + "Ġwoods": 1631, + "ĠTo": 1632, + "Ġshared": 1633, + "Ġalone": 1634, + "Ġtowards": 1635, + "cle": 1636, + "ĠMaybe": 1637, + "Ġsit": 1638, + "Ġnap": 1639, + "Ġcolorful": 1640, + "amp": 1641, + "hone": 1642, + "Ġvisit": 1643, + "igg": 1644, + "ugg": 1645, + "thy": 1646, + "ence": 1647, + "Ġtail": 1648, + "ctor": 1649, + "Ġmagical": 1650, + "Ġriver": 1651, + "Ġhig": 1652, + "Ġbelie": 1653, + "Ġgrate": 1654, + "ually": 1655, + "ang": 1656, + "ott": 1657, + "Ġhide": 1658, + "Ġsilly": 1659, + "Ġsmiles": 1660, + "ower": 1661, + "Ġside": 1662, + "Max": 1663, + "Ġroll": 1664, + "Ġgu": 1665, + "ĠAmy": 1666, + "Ġpro": 1667, + "eter": 1668, + "Ġfaster": 1669, + "Ġgrateful": 1670, + "Ġtreasure": 1671, + "ased": 1672, + "Ġsnack": 1673, + "up": 1674, + "Ġland": 1675, + "avy": 1676, + "Ġmoral": 1677, + "ained": 1678, + "Ġsister": 1679, + "ĠGra": 1680, + "Ġshap": 1681, + "Ġpaint": 1682, + "Ġslowly": 1683, + "Ġfield": 1684, + "Ġpet": 1685, + "Ġsec": 1686, + "orm": 1687, + "ĠJim": 1688, + "Ġbran": 1689, + "Ġmin": 1690, + "Ġmust": 1691, + "Ġhelping": 1692, + "Ġworried": 1693, + "Ġjob": 1694, + "Ġfox": 1695, + "set": 1696, + "Ġmoney": 1697, + "Ġpot": 1698, + "Ġqui": 1699, + "Ġbel": 1700, + "Ġacc": 1701, + "Ġliving": 1702, + "Ġmonster": 1703, + "less": 1704, + "Ġparty": 1705, + "ear": 1706, + "Ġmost": 1707, + "Ġwear": 1708, + "Ġremember": 1709, + "ire": 1710, + "Ġsign": 1711, + "wl": 1712, + "Ġcloud": 1713, + "Ġcandy": 1714, + "Ġhigher": 1715, + "ipped": 1716, + "ident": 1717, + "ĠAll": 1718, + "der": 1719, + "ze": 1720, + "Ġquiet": 1721, + "Ġtown": 1722, + "unch": 1723, + "Ġprincess": 1724, + "Ġruns": 1725, + "asket": 1726, + "em": 1727, + "Ġeverywhere": 1728, + "Ġjoin": 1729, + "Ġstepped": 1730, + "Ġbasket": 1731, + "Ġlake": 1732, + "Ġcup": 1733, + "sw": 1734, + "outh": 1735, + "Ġcarrot": 1736, + "Ġcoll": 1737, + "Ġplant": 1738, + "Ġscream": 1739, + "ĠDaddy": 1740, + "Ġbuck": 1741, + "Ġheavy": 1742, + "Ġbigger": 1743, + "Ġcho": 1744, + "Ġblank": 1745, + "Ġclosed": 1746, + "Ġwon": 1747, + "ises": 1748, + "Ġansw": 1749, + "br": 1750, + "Ġblanket": 1751, + "Ġmouth": 1752, + "ames": 1753, + "eces": 1754, + "Ġeating": 1755, + "Ġpieces": 1756, + "cket": 1757, + "Ġstre": 1758, + "Ġwelc": 1759, + "Ġwouldn": 1760, + "Ġreach": 1761, + "Ġbroke": 1762, + "eared": 1763, + "Ġdangerous": 1764, + "gs": 1765, + "ph": 1766, + "ĠMolly": 1767, + "Ġce": 1768, + "ĠMr": 1769, + "ord": 1770, + "fort": 1771, + "Ġdolls": 1772, + "ming": 1773, + "Ġdance": 1774, + "Ġdragon": 1775, + "Ġthinks": 1776, + "Ġasks": 1777, + "Ġacross": 1778, + "Okay": 1779, + "Ġchair": 1780, + "Ġteac": 1781, + "Ġbott": 1782, + "Hi": 1783, + "Ġpa": 1784, + "Ġneigh": 1785, + "Ġbar": 1786, + "Ġbike": 1787, + "Ġpack": 1788, + "iting": 1789, + "Ġdisapp": 1790, + "Ġneighb": 1791, + "Ġstuck": 1792, + "de": 1793, + "Ġgif": 1794, + "Ġfruit": 1795, + "Ġbelieve": 1796, + "cy": 1797, + "Ġve": 1798, + "Ġcrying": 1799, + "ier": 1800, + "phant": 1801, + "Ġsuddenly": 1802, + "Ġsoup": 1803, + "Ġrace": 1804, + "Ġthrew": 1805, + "Sure": 1806, + "Ġfarmer": 1807, + "Ġaccident": 1808, + "Ġdirty": 1809, + "Ġelephant": 1810, + "Ġmonkey": 1811, + "Ġheal": 1812, + "weet": 1813, + "Ġrem": 1814, + "Ġpers": 1815, + "Ġsame": 1816, + "ored": 1817, + "ventually": 1818, + "ead": 1819, + "pping": 1820, + "Ġgentle": 1821, + "Ġfree": 1822, + "Ġbrown": 1823, + "Maybe": 1824, + "Ġinst": 1825, + "Ġbee": 1826, + "ix": 1827, + "Ġben": 1828, + "Ġwin": 1829, + "Ġblack": 1830, + "Ġalong": 1831, + "Ġlonger": 1832, + "ible": 1833, + "Ġspr": 1834, + "Ġclapped": 1835, + "Finally": 1836, + "Why": 1837, + "cked": 1838, + "Ġwords": 1839, + "ĠFl": 1840, + "Ġteacher": 1841, + "Ġupset": 1842, + "chool": 1843, + "Ġpiece": 1844, + "Ġwell": 1845, + "Ġcheered": 1846, + "Ġhit": 1847, + "Ġsang": 1848, + "ment": 1849, + "Ġbowl": 1850, + "Ġbite": 1851, + "Ġdoctor": 1852, + "ĠJill": 1853, + "Ġbucket": 1854, + "Mia": 1855, + "Ġbath": 1856, + "Ġcaught": 1857, + "Ġheart": 1858, + "Ġforget": 1859, + "Ġmark": 1860, + "Ġbutt": 1861, + "Ġdry": 1862, + "Ġyoung": 1863, + "Ġsea": 1864, + "Ġcour": 1865, + "Ġdropped": 1866, + "ese": 1867, + "Ġyard": 1868, + "Ġplac": 1869, + "ourn": 1870, + "Ġschool": 1871, + "Ġswings": 1872, + "bby": 1873, + "Ġwings": 1874, + "bs": 1875, + "Ġexplained": 1876, + "Ġjourn": 1877, + "Ġbring": 1878, + "Ġdrink": 1879, + "Ġstreet": 1880, + "Ġjuice": 1881, + "ung": 1882, + "Ġnearby": 1883, + "ĠEmma": 1884, + "Ġsmo": 1885, + "Ġsmart": 1886, + "Ġpushed": 1887, + "Ġstories": 1888, + "Ġdrove": 1889, + "Ġtiny": 1890, + "Ġpen": 1891, + "Ġcourse": 1892, + "Ġes": 1893, + "Ġmine": 1894, + "Ġtoday": 1895, + "Ġpocket": 1896, + "Ġyear": 1897, + "aughter": 1898, + "Ġje": 1899, + "Ġsecret": 1900, + "Ġexp": 1901, + "Ġwithout": 1902, + "Ġsinging": 1903, + "Ġwelcome": 1904, + "Ġhappen": 1905, + "to": 1906, + "Ġfit": 1907, + "Ġstu": 1908, + "vel": 1909, + "ract": 1910, + "Ġbus": 1911, + "Ġcheese": 1912, + "Ġcray": 1913, + "Ġfairy": 1914, + "Ġjar": 1915, + "Ġturns": 1916, + "Ġbush": 1917, + "bow": 1918, + "llo": 1919, + "Ġexploring": 1920, + "Ġlon": 1921, + "Ġstand": 1922, + "itting": 1923, + "Ġseemed": 1924, + "Ġspoon": 1925, + "ĠFinally": 1926, + "Ġgames": 1927, + "Ġwoke": 1928, + "issed": 1929, + "Ġresp": 1930, + "Ġtrou": 1931, + "Ġtwins": 1932, + "Ġhoped": 1933, + "Ġatt": 1934, + "Ġlonely": 1935, + "Ġant": 1936, + "ĠMama": 1937, + "orrow": 1938, + "Ġnoises": 1939, + "Ġhugs": 1940, + "ountain": 1941, + "yard": 1942, + "Ġdaughter": 1943, + "Ġinv": 1944, + "Ġkey": 1945, + "ail": 1946, + "Ġrocks": 1947, + "Ġdisco": 1948, + "Ġsu": 1949, + "Ġlunch": 1950, + "read": 1951, + "oun": 1952, + "vered": 1953, + "Ġbackyard": 1954, + "Ġsuc": 1955, + "Ġcow": 1956, + "Ġapple": 1957, + "ek": 1958, + "Ġswam": 1959, + "Ġshoes": 1960, + "Ġstars": 1961, + "Ġshook": 1962, + "ittens": 1963, + "Ġmiss": 1964, + "ches": 1965, + "Ġwish": 1966, + "Ġrelie": 1967, + "fused": 1968, + "Ġshapes": 1969, + "urple": 1970, + "Ġtower": 1971, + "ity": 1972, + "Ġcorn": 1973, + "Ġmoved": 1974, + "shine": 1975, + "Ġ3": 1976, + "Ġthough": 1977, + "Ġcir": 1978, + "umb": 1979, + "Ġtaking": 1980, + "ts": 1981, + "ĠTweet": 1982, + "ape": 1983, + "Ġrainbow": 1984, + "Ġthrow": 1985, + "Ġstar": 1986, + "Ġbench": 1987, + "Ġneck": 1988, + "ocked": 1989, + "Ġcreature": 1990, + "Ġbubble": 1991, + "Ġlate": 1992, + "Ġadventures": 1993, + "Ġowl": 1994, + "ions": 1995, + "And": 1996, + "Ġwhe": 1997, + "Ġpar": 1998, + "Ġshell": 1999, + "Ġkite": 2000, + "ĠFluffy": 2001, + "ining": 2002, + "Ġmil": 2003, + "unt": 2004, + "Ġfence": 2005, + "Ġmix": 2006, + "Ġlift": 2007, + "Ġaccidentally": 2008, + "Ġwise": 2009, + "Ġhello": 2010, + "Ġpurple": 2011, + "pper": 2012, + "Ġonto": 2013, + "Ġspin": 2014, + "ten": 2015, + "ours": 2016, + "Ġsitting": 2017, + "idge": 2018, + "Ġsunshine": 2019, + "Ġcute": 2020, + "Ġmat": 2021, + "ward": 2022, + "Ġshop": 2023, + "Ġwhist": 2024, + "Ġdist": 2025, + "Ġorange": 2026, + "Lila": 2027, + "Ġnaught": 2028, + "ĠSammy": 2029, + "Ġnaughty": 2030, + "Ġnest": 2031, + "ose": 2032, + "ors": 2033, + "lla": 2034, + "Ġmilk": 2035, + "fish": 2036, + "Ġaf": 2037, + "Ġboun": 2038, + "Ġphone": 2039, + "Ġarms": 2040, + "Ġeas": 2041, + "ness": 2042, + "Ġcarry": 2043, + "ĠGrandma": 2044, + "og": 2045, + "Ġbought": 2046, + "Ġdriver": 2047, + "Ġinstead": 2048, + "Later": 2049, + "Ġtalked": 2050, + "Ġsearched": 2051, + "gged": 2052, + "Ġpig": 2053, + "Ġcozy": 2054, + "Ġsunny": 2055, + "itty": 2056, + "Ġfeels": 2057, + "Ġfan": 2058, + "Ġwondered": 2059, + "Ġcomes": 2060, + "Ġswimming": 2061, + "uched": 2062, + "Ġtouched": 2063, + "Ġpop": 2064, + "ĠInside": 2065, + "ale": 2066, + "Ġlucky": 2067, + "Ġlaughing": 2068, + "Ġbelong": 2069, + "Ġsounds": 2070, + "Ġwom": 2071, + "Ġcave": 2072, + "let": 2073, + "Ġjourney": 2074, + "Ġwoman": 2075, + "Ġcou": 2076, + "Ġnothing": 2077, + "Ġcoat": 2078, + "Ġrelieved": 2079, + "ĠO": 2080, + "Ġpeace": 2081, + "adow": 2082, + "Ġnose": 2083, + "Ġbusy": 2084, + "Bob": 2085, + "Ġpast": 2086, + "Ġapples": 2087, + "Ġsick": 2088, + "Ġrope": 2089, + "Ġpatient": 2090, + "Ġact": 2091, + "Ġpudd": 2092, + "Ġinc": 2093, + "raid": 2094, + "Ġwatching": 2095, + "Ġbranch": 2096, + "Ġafraid": 2097, + "Ġtom": 2098, + "ider": 2099, + "Ġdays": 2100, + "Ġret": 2101, + "cing": 2102, + "ĠMary": 2103, + "Ġband": 2104, + "Ġfarm": 2105, + "Ġhuge": 2106, + "ĠThank": 2107, + "ator": 2108, + "Ġexciting": 2109, + "Ġdanced": 2110, + "thday": 2111, + "Ġwar": 2112, + "ĠSoon": 2113, + "ball": 2114, + "Ġgigg": 2115, + "blem": 2116, + "Ġzoom": 2117, + "Ġmad": 2118, + "Ġvill": 2119, + "Ġforever": 2120, + "Ġproblem": 2121, + "Ġcollect": 2122, + "ped": 2123, + "irt": 2124, + "Ġminut": 2125, + "Ġwra": 2126, + "ho": 2127, + "Ġlife": 2128, + "Ġknee": 2129, + "ces": 2130, + "Do": 2131, + "cer": 2132, + "mp": 2133, + "arp": 2134, + "Ġking": 2135, + "Ġasleep": 2136, + "Ġreturn": 2137, + "Ġsharp": 2138, + "Ġrel": 2139, + "Ġpower": 2140, + "eth": 2141, + "Ġholding": 2142, + "Ġfing": 2143, + "Ġhealthy": 2144, + "ĠMittens": 2145, + "Ġprot": 2146, + "Ġmeet": 2147, + "Ġorgan": 2148, + "Ġclever": 2149, + "Ġspotted": 2150, + "Ġletter": 2151, + "Ġgather": 2152, + "alm": 2153, + "Of": 2154, + "ere": 2155, + "Ġround": 2156, + "Ġstorm": 2157, + "Ġprotect": 2158, + "Ġgift": 2159, + "amed": 2160, + "Ġmail": 2161, + "ĠJen": 2162, + "Ġbirthday": 2163, + "Ġpres": 2164, + "Ġsaf": 2165, + "Ġneighbor": 2166, + "epend": 2167, + "Ġtakes": 2168, + "sc": 2169, + "Ġear": 2170, + "Ġcomfort": 2171, + "Ġent": 2172, + "Ġrep": 2173, + "Ġperson": 2174, + "Ġsmiling": 2175, + "ĠWith": 2176, + "Ġgently": 2177, + "Ġpointed": 2178, + "Every": 2179, + "owed": 2180, + "Ġspo": 2181, + "col": 2182, + "Ġtasty": 2183, + "ular": 2184, + "ĠJimmy": 2185, + "Ġplane": 2186, + "sting": 2187, + "hin": 2188, + "Ġhoney": 2189, + "itch": 2190, + "Ġpill": 2191, + "Ġpass": 2192, + "itten": 2193, + "Ġmatter": 2194, + "Ġadm": 2195, + "Ġtasted": 2196, + "Ġsmelled": 2197, + "nic": 2198, + "ove": 2199, + "Ġeag": 2200, + "Ġshining": 2201, + "Hey": 2202, + "Ġhope": 2203, + "Ġplaces": 2204, + "Ġkick": 2205, + "ĠCh": 2206, + "Ġpicnic": 2207, + "Ġbottle": 2208, + "Ġrespect": 2209, + "Ġblew": 2210, + "Ġappeared": 2211, + "Ġsupp": 2212, + "Tommy": 2213, + "ĠCome": 2214, + "Ġspider": 2215, + "Ġintere": 2216, + "Ġcheer": 2217, + "board": 2218, + "Ġwearing": 2219, + "Ġpresent": 2220, + "ert": 2221, + "Ġmist": 2222, + "ize": 2223, + "Ġtrouble": 2224, + "Ġtrip": 2225, + "Ġpract": 2226, + "Ġkiss": 2227, + "Ġquest": 2228, + "ond": 2229, + "Ġash": 2230, + "Ġloves": 2231, + "Ġtightly": 2232, + "gg": 2233, + "arge": 2234, + "Ġsur": 2235, + "Ġspread": 2236, + "aur": 2237, + "Ġwide": 2238, + "ones": 2239, + "umber": 2240, + "Ġfeather": 2241, + "Ġsplas": 2242, + "ont": 2243, + "ndp": 2244, + "Ġlater": 2245, + "Ġwhole": 2246, + "Ġanyone": 2247, + "Ġbread": 2248, + "Molly": 2249, + "Ġdes": 2250, + "Ġrec": 2251, + "ella": 2252, + "Ġroad": 2253, + "In": 2254, + "Ġoce": 2255, + "Ġgiant": 2256, + "Ġocean": 2257, + "els": 2258, + "Ġpuzz": 2259, + "Ġcookie": 2260, + "Ġpan": 2261, + "Ġpath": 2262, + "ÅĵI": 2263, + "Ġconfused": 2264, + "Ġscreamed": 2265, + "Ġclouds": 2266, + "eb": 2267, + "Ġimpress": 2268, + "Ġfront": 2269, + "ĠGo": 2270, + "Ġseat": 2271, + "lex": 2272, + "Ġlay": 2273, + "Ġwhich": 2274, + "la": 2275, + "Ġthin": 2276, + "Ġteach": 2277, + "erly": 2278, + "Ġdi": 2279, + "Ġanswer": 2280, + "ungle": 2281, + "Ġap": 2282, + "ssed": 2283, + "hile": 2284, + "ub": 2285, + "Ġlarge": 2286, + "Ġjungle": 2287, + "Ġpromise": 2288, + "Ġser": 2289, + "Ġber": 2290, + "Ġwild": 2291, + "Ġbark": 2292, + "Ġmach": 2293, + "oose": 2294, + "Ġbutton": 2295, + "ndpa": 2296, + "Good": 2297, + "Ġmyster": 2298, + "Ġsnake": 2299, + "Ġvillage": 2300, + "Ġhor": 2301, + "ĠEmily": 2302, + "Come": 2303, + "Ġtool": 2304, + "Ġputs": 2305, + "Ġlast": 2306, + "Ġimag": 2307, + "ĠJake": 2308, + "Lucy": 2309, + "Ġtid": 2310, + "Ġban": 2311, + "lebr": 2312, + "Ġeventually": 2313, + "Ġslid": 2314, + "Ġwave": 2315, + "Ġlive": 2316, + "Ġchased": 2317, + "ĠEverywhere": 2318, + "Ġcelebr": 2319, + "Ġcrayons": 2320, + "Just": 2321, + "Ġspent": 2322, + "Ġwrite": 2323, + "Ġgives": 2324, + "unk": 2325, + "ife": 2326, + "Ġdecor": 2327, + "Ġfoot": 2328, + "Ġdelight": 2329, + "Ġsol": 2330, + "Ġsandw": 2331, + "Ġdirt": 2332, + "ĠJenny": 2333, + "Ġtalking": 2334, + "Ġtreats": 2335, + "Ġcart": 2336, + "ching": 2337, + "Ġwaiting": 2338, + "Ġhiding": 2339, + "Ġsnowman": 2340, + "Ġinteresting": 2341, + "Ġhon": 2342, + "ony": 2343, + "Ġshelf": 2344, + "Ġtaste": 2345, + "een": 2346, + "Jim": 2347, + "Ġstep": 2348, + "Mum": 2349, + "Ġhelpful": 2350, + "Ġfeet": 2351, + "Ġshy": 2352, + "Ġgoes": 2353, + "Ġval": 2354, + "Ġemb": 2355, + "Ġstood": 2356, + "Ġshr": 2357, + "Ġbuilding": 2358, + "Ġoven": 2359, + "Ġstret": 2360, + "Ġlovely": 2361, + "Ġducks": 2362, + "Ġcorner": 2363, + "Ġwash": 2364, + "cks": 2365, + "Ġspe": 2366, + "Ġpretended": 2367, + "Ġrolled": 2368, + "Ġrude": 2369, + "old": 2370, + "Ġcalm": 2371, + "Ġpile": 2372, + "Ġteeth": 2373, + "Ġjumping": 2374, + "Ġzoo": 2375, + "Ġpuddle": 2376, + "Ġtidy": 2377, + "Mama": 2378, + "iger": 2379, + "ipe": 2380, + "Ġbugs": 2381, + "Ġashamed": 2382, + "Ġbat": 2383, + "ina": 2384, + "ĠRe": 2385, + "Ġob": 2386, + "Ġstra": 2387, + "ĠSt": 2388, + "Ġwhistle": 2389, + "Ġbanan": 2390, + "not": 2391, + "Ġnet": 2392, + "Ġforgive": 2393, + "Ġprince": 2394, + "ener": 2395, + "Ġshirt": 2396, + "Ġsel": 2397, + "Ġcoo": 2398, + "Ġembar": 2399, + "umpy": 2400, + "Ġsparkly": 2401, + "Ġremind": 2402, + "Ġsug": 2403, + "Ġcross": 2404, + "Ġmind": 2405, + "Ġcannot": 2406, + "Ġsuccess": 2407, + "Ġembarra": 2408, + "Soon": 2409, + "Ġmot": 2410, + "Ġsharing": 2411, + "Ġworm": 2412, + "Ġfold": 2413, + "Ġpeaceful": 2414, + "Ġwore": 2415, + "Ġmountain": 2416, + "Ġblow": 2417, + "lice": 2418, + "Ġign": 2419, + "Ġbell": 2420, + "Ġfurry": 2421, + "Ġgrab": 2422, + "als": 2423, + "ometimes": 2424, + "Ġwhenever": 2425, + "Ġpour": 2426, + "vous": 2427, + "llie": 2428, + "Ġpush": 2429, + "Ġbreath": 2430, + "Ġmachine": 2431, + "Ġsugar": 2432, + "Ġchick": 2433, + "Ġcreative": 2434, + "St": 2435, + "ned": 2436, + "Ġbal": 2437, + "Ġstir": 2438, + "Ġdolp": 2439, + "Ġtries": 2440, + "Now": 2441, + "Ġlad": 2442, + "ĠLe": 2443, + "Ġmed": 2444, + "Ġpool": 2445, + "Ġgener": 2446, + "Ġsal": 2447, + "ork": 2448, + "ult": 2449, + "Ġspray": 2450, + "Ġship": 2451, + "ian": 2452, + "Ġhum": 2453, + "Ġhel": 2454, + "isa": 2455, + "Ġselfish": 2456, + "Ġhar": 2457, + "Ġfaces": 2458, + "Ġmessy": 2459, + "Ġ'": 2460, + "Ġple": 2461, + "ĠSome": 2462, + "ĠHow": 2463, + "Ġgold": 2464, + "Ġchocol": 2465, + "Please": 2466, + "Ġminutes": 2467, + "Ġpillow": 2468, + "My": 2469, + "aking": 2470, + "Ġtun": 2471, + "Ġmissed": 2472, + "ighed": 2473, + "Ġvan": 2474, + "Ġfixed": 2475, + "Ġfighting": 2476, + "oin": 2477, + "Ġfig": 2478, + "Ġembarrassed": 2479, + "Ġmeant": 2480, + "Ġdrive": 2481, + "Ġtomorrow": 2482, + "Ġhours": 2483, + "Ġca": 2484, + "Ġner": 2485, + "Ġspicy": 2486, + "Ġknowing": 2487, + "ĠPlease": 2488, + "Sarah": 2489, + "Ġladder": 2490, + "ital": 2491, + "Ġhorse": 2492, + "ato": 2493, + "Ġug": 2494, + "ĠJust": 2495, + "Ġtrust": 2496, + "Ġhosp": 2497, + "Ġugly": 2498, + "Ġhospital": 2499, + "Ġchew": 2500, + "Ġbedroom": 2501, + "Ġeasy": 2502, + "Ġnervous": 2503, + "Ġrob": 2504, + "Ġthankful": 2505, + "Ġcarrots": 2506, + "Dad": 2507, + "elly": 2508, + "Ġgrew": 2509, + "Ġcolour": 2510, + "Ġbarked": 2511, + "Ġcouch": 2512, + "Ġnut": 2513, + "Ġmeadow": 2514, + "Ġtaught": 2515, + "Ġunderstood": 2516, + "Ġdisappoin": 2517, + "Ġpas": 2518, + "ano": 2519, + "Ġoften": 2520, + "Ġclass": 2521, + "Ġpassed": 2522, + "Ġknocked": 2523, + "Ġlanded": 2524, + "Ġmic": 2525, + "ird": 2526, + "Ġfear": 2527, + "Ġcoin": 2528, + "Ġmud": 2529, + "work": 2530, + "Ġdeter": 2531, + "Ġreg": 2532, + "Ġreturned": 2533, + "Ġdeterm": 2534, + "Ġhur": 2535, + "Ġfinish": 2536, + "izz": 2537, + "Ġtea": 2538, + "Ġsmooth": 2539, + "mo": 2540, + "Ġstuff": 2541, + "Ġmov": 2542, + "Ġgenerous": 2543, + "To": 2544, + "ĠOn": 2545, + "isp": 2546, + "Ġdrum": 2547, + "obby": 2548, + "Ġbubbles": 2549, + "Ġrobot": 2550, + "Ġthese": 2551, + "Ġseed": 2552, + "Ġglass": 2553, + "Ġdisappointed": 2554, + "ground": 2555, + "Ġring": 2556, + "Ġcard": 2557, + "Ġhears": 2558, + "Ġcarried": 2559, + "Ġmoon": 2560, + "Ġchocolate": 2561, + "Ġleaf": 2562, + "Ġsnacks": 2563, + "Ġcirc": 2564, + "Ġfancy": 2565, + "dle": 2566, + "Ġspend": 2567, + "ĠAnn": 2568, + "ustr": 2569, + "Ġcrab": 2570, + "Ġsongs": 2571, + "pack": 2572, + "Ġpir": 2573, + "ecially": 2574, + "osaur": 2575, + "Ġsold": 2576, + "Ġbridge": 2577, + "ĠEven": 2578, + "Ġsleepy": 2579, + "Ġhonest": 2580, + "Ġmir": 2581, + "ĠBr": 2582, + "Ġpaw": 2583, + "pecially": 2584, + "Ġav": 2585, + "ĠWhy": 2586, + "izzy": 2587, + "Ġchest": 2588, + "Ġwolf": 2589, + "Ġdinosaur": 2590, + "Ġespecially": 2591, + "Ġmap": 2592, + "Ġgas": 2593, + "ory": 2594, + "Ġchange": 2595, + "Ġgroup": 2596, + "Ġgathered": 2597, + "Ġsour": 2598, + "Ġpoor": 2599, + "Ġsplash": 2600, + "Ġwaves": 2601, + "Ġforward": 2602, + "Ġclown": 2603, + "Ġmysterious": 2604, + "ror": 2605, + "Ġbody": 2606, + "Ġfrustr": 2607, + "Ġcraw": 2608, + "Ġfake": 2609, + "Ġpin": 2610, + "Ġok": 2611, + "Ġrad": 2612, + "Ġwhisp": 2613, + "Ġtwirl": 2614, + "Ġstring": 2615, + "Ġsmelly": 2616, + "ĠTogether": 2617, + "sist": 2618, + "Ġcand": 2619, + "Ġgate": 2620, + "Ġscar": 2621, + "ĠKitty": 2622, + "Ġneckl": 2623, + "Billy": 2624, + "Ġmotor": 2625, + "ying": 2626, + "ote": 2627, + "Ġshore": 2628, + "urse": 2629, + "Ġjew": 2630, + "Ġweak": 2631, + "Ġgrumpy": 2632, + "Ġfinger": 2633, + "Ġsack": 2634, + "Ġkitten": 2635, + "Ġpolite": 2636, + "Ġglue": 2637, + "uce": 2638, + "Ġnumber": 2639, + "airs": 2640, + "Ġesc": 2641, + "Ġmirror": 2642, + "Ġsheep": 2643, + "Ġvase": 2644, + "Ġplants": 2645, + "Ġberries": 2646, + "most": 2647, + "Be": 2648, + "den": 2649, + "Ġjelly": 2650, + "ĠTheir": 2651, + "Ġemp": 2652, + "itter": 2653, + "There": 2654, + "Ġscr": 2655, + "Ġgray": 2656, + "Ġpuzzle": 2657, + "Ġnecklace": 2658, + "Ġmaybe": 2659, + "Ġplate": 2660, + "Ġalmost": 2661, + "usie": 2662, + "Ġfrustrated": 2663, + "Ġempty": 2664, + "port": 2665, + "orry": 2666, + "Ġthick": 2667, + "Ġjam": 2668, + "Ġskipped": 2669, + "ensive": 2670, + "ĠTeddy": 2671, + "aged": 2672, + "Ġcloset": 2673, + "Ġhidden": 2674, + "Ġexpensive": 2675, + "Ġdetermined": 2676, + "Ġbored": 2677, + "Ġlit": 2678, + "Ġdrew": 2679, + "Ġlegs": 2680, + "Ġhopping": 2681, + "ates": 2682, + "ĠSometimes": 2683, + "Ġsticks": 2684, + "Ġones": 2685, + "ĠAt": 2686, + "Ġbirdie": 2687, + "ĠDon": 2688, + "Ġpolice": 2689, + "Ġfruits": 2690, + "Ġtick": 2691, + "Ġaunt": 2692, + "Ġwis": 2693, + "Ġbake": 2694, + "Ġline": 2695, + "ots": 2696, + "Ġstone": 2697, + "Ġdogs": 2698, + "aul": 2699, + "Ġfigure": 2700, + "Jane": 2701, + "Ġsight": 2702, + "Ġcries": 2703, + "ĠBe": 2704, + "Ġmoving": 2705, + "Ġcontent": 2706, + "ĠBrown": 2707, + "Ġsaved": 2708, + "Ġcloth": 2709, + "?\".": 2710, + "box": 2711, + "Ġbranches": 2712, + "Ġpacked": 2713, + "Ġpowerful": 2714, + "Ġsplashed": 2715, + "Ġfeed": 2716, + "ested": 2717, + "Ġyelled": 2718, + "Where": 2719, + "dge": 2720, + "kin": 2721, + "Ġtiger": 2722, + "Ġpay": 2723, + "iddle": 2724, + "oop": 2725, + "ĠBella": 2726, + "Ġexam": 2727, + "Ġtaken": 2728, + "Ġcrane": 2729, + "Ġspoil": 2730, + "'d": 2731, + "Ġhang": 2732, + "Ġflag": 2733, + "Ġanswered": 2734, + "Ġdiscovered": 2735, + "ĠLeo": 2736, + "ush": 2737, + "Ġsave": 2738, + "Ġseek": 2739, + "Ġexcite": 2740, + "Ġpray": 2741, + "Ġjog": 2742, + "Ġstanding": 2743, + "Ġdolphin": 2744, + "Ġjack": 2745, + "Are": 2746, + "era": 2747, + "rag": 2748, + "Ġflut": 2749, + "getable": 2750, + "Ġboring": 2751, + "Ġlock": 2752, + "Ġincred": 2753, + "Ġexcitement": 2754, + "Ġsighed": 2755, + "Ġwip": 2756, + "Ġfis": 2757, + "Ġfill": 2758, + "Ġsoap": 2759, + "ĠMum": 2760, + "ola": 2761, + "Ġpleased": 2762, + "Ġjacket": 2763, + "light": 2764, + "Ġsugg": 2765, + "Ġmiddle": 2766, + "Ġeld": 2767, + "Ġwrote": 2768, + "Ġballoons": 2769, + "Ġorganized": 2770, + "que": 2771, + "Ġsailed": 2772, + "Their": 2773, + "Ġtells": 2774, + "Ġoy": 2775, + "Ġstones": 2776, + "Ġboard": 2777, + "Ġspun": 2778, + "Ġvegetable": 2779, + "lies": 2780, + "Ġind": 2781, + "iness": 2782, + "Ġworking": 2783, + "lower": 2784, + "alk": 2785, + "iff": 2786, + "Ġparrot": 2787, + "Stop": 2788, + "Ġscarf": 2789, + "Ġwagged": 2790, + "Ġclap": 2791, + "Ġmeasure": 2792, + "asing": 2793, + "Ġtears": 2794, + "body": 2795, + "Ġcomfortable": 2796, + "Ġleg": 2797, + "Ġswan": 2798, + "Ġtrue": 2799, + "Ġfright": 2800, + "Ġdrop": 2801, + "Ġsmoke": 2802, + "Ġfeathers": 2803, + "Ġincredible": 2804, + "ggs": 2805, + "ual": 2806, + "Ġtie": 2807, + "Ġcap": 2808, + "Ġlose": 2809, + "Ġpus": 2810, + "Ġeager": 2811, + "Ġmovie": 2812, + "fic": 2813, + "Ġcop": 2814, + "Ġeggs": 2815, + "oof": 2816, + "Ġrid": 2817, + "Ġcovered": 2818, + "Ġfil": 2819, + "Ġlem": 2820, + "Ġgrap": 2821, + "Ġdisappeared": 2822, + "Ġrecord": 2823, + "ĠTV": 2824, + "ĠBl": 2825, + "Ġext": 2826, + "Ġord": 2827, + "Ġgiggled": 2828, + "corn": 2829, + "Ġpri": 2830, + "Ġham": 2831, + "Ġrare": 2832, + "Ġyourself": 2833, + "Ġeng": 2834, + "Ġwhale": 2835, + "Ġknife": 2836, + "ĠLisa": 2837, + "Ġteam": 2838, + "ĠJohnny": 2839, + "zed": 2840, + "Ġtest": 2841, + "Ġbone": 2842, + "Ġdizzy": 2843, + "Ġlibr": 2844, + "Ġwrap": 2845, + "Ġquestions": 2846, + "iv": 2847, + "Ġlying": 2848, + "Ġstamp": 2849, + "Ġlicked": 2850, + "Ġnoisy": 2851, + "Ġkindness": 2852, + "Ġdiffic": 2853, + "Ġdrawing": 2854, + "apping": 2855, + "Jimmy": 2856, + "Ġstraw": 2857, + "Ġburn": 2858, + "Ġwagon": 2859, + "ately": 2860, + "Ġgrown": 2861, + "Ġrocket": 2862, + "Ġpopcorn": 2863, + "Sally": 2864, + "Ġbatter": 2865, + "Ġwand": 2866, + "Ġanc": 2867, + "ĠEnd": 2868, + "Ġsword": 2869, + "Ġhappening": 2870, + "Ġlifted": 2871, + "Ġbalance": 2872, + "Ġextra": 2873, + "Sorry": 2874, + "ha": 2875, + "Ġwander": 2876, + "Ġfat": 2877, + "ĠMark": 2878, + "ĠElla": 2879, + "Ġbecome": 2880, + "Ġtrack": 2881, + "Ġsmaller": 2882, + "Ġmistake": 2883, + "Ġtoast": 2884, + "ating": 2885, + "Ġmanaged": 2886, + "selves": 2887, + "ĠPeter": 2888, + "Ġbathroom": 2889, + "Ġharm": 2890, + "Ġdifficult": 2891, + "Ġtwist": 2892, + "Ġoffered": 2893, + "Ġruined": 2894, + "Ġmatch": 2895, + "Ġancient": 2896, + "Ġgar": 2897, + "illi": 2898, + "Ġdistance": 2899, + "Ġplayground": 2900, + "Ġlazy": 2901, + "Ġdancing": 2902, + "Ġadventur": 2903, + "af": 2904, + "ĠBa": 2905, + "Ġcalling": 2906, + "Ġcleaned": 2907, + "Ġcrown": 2908, + "ormal": 2909, + "Ġonce": 2910, + "Ġboss": 2911, + "Ġsell": 2912, + "Ġwalks": 2913, + "Ġfrightened": 2914, + "erri": 2915, + "Ġmask": 2916, + "Ġfluffy": 2917, + "emo": 2918, + "ĠZ": 2919, + "Ġfrag": 2920, + "Ġdest": 2921, + "Ġnormal": 2922, + "Ġyog": 2923, + "Ġpicks": 2924, + "Ġeye": 2925, + "Ġdrank": 2926, + "Ġcircle": 2927, + "Ġadventurous": 2928, + "Spot": 2929, + "Ġtent": 2930, + "Ġpress": 2931, + "Ġkissed": 2932, + "ict": 2933, + "Ġsciss": 2934, + "Ġgets": 2935, + "Ġtrunk": 2936, + "Ġcheck": 2937, + "Ġenjoying": 2938, + "Ġshells": 2939, + "Ġelderly": 2940, + "Ġscissors": 2941, + "Ġterri": 2942, + "Ġtoug": 2943, + "ages": 2944, + "Ġsteal": 2945, + "Me": 2946, + "ready": 2947, + "ete": 2948, + "Ġalready": 2949, + "ĠLittle": 2950, + "Ġjuicy": 2951, + "Ġstudy": 2952, + "Ġhorn": 2953, + "erp": 2954, + "Ġpizz": 2955, + "Ġallig": 2956, + "Ġlouder": 2957, + "Ġtowel": 2958, + "Ġcircles": 2959, + "Ġlead": 2960, + "Ġroar": 2961, + "Ġrose": 2962, + "Ġslipped": 2963, + "Ġtele": 2964, + "Ġfamous": 2965, + "Ġarg": 2966, + "Ġcheerful": 2967, + "ĠRex": 2968, + "Ġtough": 2969, + "ipper": 2970, + "Ġbarn": 2971, + "Ġfool": 2972, + "Ġdig": 2973, + "Ġdish": 2974, + "Ġagainst": 2975, + "Ġdeer": 2976, + "Ġsweetie": 2977, + "Ġengine": 2978, + "Ġed": 2979, + "Ġfather": 2980, + "Ġcover": 2981, + "Ġyours": 2982, + "Ġneeds": 2983, + "Joe": 2984, + "Ġbags": 2985, + "Ġspinning": 2986, + "Sue": 2987, + "Ġaw": 2988, + "anged": 2989, + "amond": 2990, + "Ġyarn": 2991, + "oph": 2992, + "Ġgiving": 2993, + "Ġdiamond": 2994, + "Ġsandwich": 2995, + "Ġindepend": 2996, + "Ġfragile": 2997, + "Who": 2998, + "cept": 2999, + "Ġnosy": 3000, + "Ġtwig": 3001, + "Ġhurts": 3002, + "Ġsuggested": 3003, + "Ġlemon": 3004, + "iest": 3005, + "ination": 3006, + "orge": 3007, + "Ġnod": 3008, + "Ġbloom": 3009, + "Ġhose": 3010, + "Ġboxes": 3011, + "Ġrealised": 3012, + "Ġfrown": 3013, + "ache": 3014, + "Ġrelax": 3015, + "Ġhammer": 3016, + "Ġadded": 3017, + "Ġweird": 3018, + "Ġmod": 3019, + "Ġcrack": 3020, + "Ġalligator": 3021, + "Amy": 3022, + "Gra": 3023, + "oot": 3024, + "Ġshake": 3025, + "rage": 3026, + "Ġlearning": 3027, + "Ġvegetables": 3028, + "Ġadd": 3029, + "Ġmelt": 3030, + "ĠSusie": 3031, + "elp": 3032, + "ĠBobby": 3033, + "Ġtrap": 3034, + "Ġcollar": 3035, + "Ġstubb": 3036, + "ique": 3037, + "Ġborrow": 3038, + "Ġpump": 3039, + "Ġloudly": 3040, + "Ġsearch": 3041, + "Ġchicken": 3042, + "Ġpizza": 3043, + "Ġindependent": 3044, + "Ġstubborn": 3045, + "Ġref": 3046, + "Ġasking": 3047, + "Ġbrilli": 3048, + "Ġdecide": 3049, + "Ġflash": 3050, + "Ġcaterp": 3051, + "Ġqueen": 3052, + "Ġsnee": 3053, + "ĠRose": 3054, + "Ġinvited": 3055, + "Ġstrawber": 3056, + "Ġfoolish": 3057, + "Ġcaterpill": 3058, + "gy": 3059, + "Ġtur": 3060, + "Ġdepend": 3061, + "Ġguit": 3062, + "Ġstri": 3063, + "Ġknight": 3064, + "Ġboys": 3065, + "Ġpee": 3066, + "Ġcomb": 3067, + "Ġclear": 3068, + "Ġunique": 3069, + "Ġbuttons": 3070, + "ĠTweety": 3071, + "Ġcolourful": 3072, + "Ġescape": 3073, + "Ġbrilliant": 3074, + "Ġrich": 3075, + "ĠBu": 3076, + "Ġsaying": 3077, + "Ġrode": 3078, + "Ġskin": 3079, + "Ġgarage": 3080, + "Help": 3081, + "dom": 3082, + "Ġmummy": 3083, + "Ġbeak": 3084, + "ured": 3085, + "aw": 3086, + "eorge": 3087, + "rop": 3088, + "edient": 3089, + "Ġunus": 3090, + "Ġoyster": 3091, + "Ġturkey": 3092, + "ĠOK": 3093, + "Ġfountain": 3094, + "Ġpatter": 3095, + "Ġweek": 3096, + "Ġchase": 3097, + "Ġrubb": 3098, + "Ġcounted": 3099, + "Ġregular": 3100, + "Ġwiped": 3101, + "Ġguitar": 3102, + "Ġunusual": 3103, + "land": 3104, + "Ġumb": 3105, + "Ġfork": 3106, + "Ġshows": 3107, + "Ġlights": 3108, + "Ġwalls": 3109, + "Ġdreamed": 3110, + "Ġbanana": 3111, + "Ġsuccessful": 3112, + "Ġjewel": 3113, + "Ġpattern": 3114, + "Ġumbre": 3115, + "cil": 3116, + "ray": 3117, + "inal": 3118, + "Ġlow": 3119, + "imed": 3120, + "Ġswe": 3121, + "Ġcryst": 3122, + "Ġpencil": 3123, + "Ġsafely": 3124, + "Ġtools": 3125, + "Well": 3126, + "Ġtape": 3127, + "Ġstream": 3128, + "ceed": 3129, + "Ġresc": 3130, + "Ġabove": 3131, + "Ġharder": 3132, + "Ġguess": 3133, + "Ġbottom": 3134, + "Ġmarket": 3135, + "Ġimpressed": 3136, + "Ġwaff": 3137, + "anut": 3138, + "iginal": 3139, + "Ġnotice": 3140, + "illa": 3141, + "Ġnods": 3142, + "Ġpeanut": 3143, + "olog": 3144, + "Ġpatch": 3145, + "Ġcrawled": 3146, + "Ġcaterpillar": 3147, + "na": 3148, + "Ġtank": 3149, + "Ġtask": 3150, + "Ġtire": 3151, + "Ġgun": 3152, + "Ġload": 3153, + "Ġcompet": 3154, + "umbled": 3155, + "Ġapolog": 3156, + "Ġsolve": 3157, + "How": 3158, + "Mummy": 3159, + "fast": 3160, + "hip": 3161, + "ĠNo": 3162, + "Ġbitter": 3163, + "arl": 3164, + "Ġears": 3165, + "Ġshone": 3166, + "Ġbrush": 3167, + "ĠFin": 3168, + "Ġoriginal": 3169, + "Ġfier": 3170, + "Ġglow": 3171, + "ĠNemo": 3172, + "Ġbreakfast": 3173, + "Ġadmired": 3174, + "ĠBlue": 3175, + "Ġedge": 3176, + "Ġdependable": 3177, + "und": 3178, + "uth": 3179, + "ĠIf": 3180, + "Ġclay": 3181, + "Ġstronger": 3182, + "Ġbossy": 3183, + "Ġcudd": 3184, + "aser": 3185, + "Ġgrandpa": 3186, + "Ġknows": 3187, + "Ġpicking": 3188, + "Ġjoke": 3189, + "Ġboats": 3190, + "Ġbald": 3191, + "Ġmedic": 3192, + "Ġscrat": 3193, + "Ġcone": 3194, + "Ġpenny": 3195, + "Ġlean": 3196, + "Ġraft": 3197, + "uggled": 3198, + "Ġsucceed": 3199, + "Ġobedient": 3200, + "Ġhumble": 3201, + "Ġlibrary": 3202, + "Re": 3203, + "head": 3204, + "Ġph": 3205, + "Ġnumb": 3206, + "Ġsauce": 3207, + "ĠBear": 3208, + "Ġtray": 3209, + "Ġproudly": 3210, + "Here": 3211, + "Ġsounded": 3212, + "Ġsquare": 3213, + "Ġterrible": 3214, + "Ġcrystal": 3215, + "Ġfierce": 3216, + "Jill": 3217, + "Ġwitch": 3218, + "inary": 3219, + "Ġhappiness": 3220, + "ĠLola": 3221, + "Ġhanded": 3222, + "Ġfridge": 3223, + "fa": 3224, + "Ġrough": 3225, + "ention": 3226, + "Ġchanged": 3227, + "Ġjoined": 3228, + "Ġraven": 3229, + "Ġmighty": 3230, + "Ġchose": 3231, + "Ġwheel": 3232, + "Ġumbrella": 3233, + "Ġtag": 3234, + "Ġcri": 3235, + "Ġow": 3236, + "arry": 3237, + "imp": 3238, + "ĠBill": 3239, + "Ġhelps": 3240, + "Ġfinding": 3241, + "Ġcleaning": 3242, + "Ġsquee": 3243, + "Ġfirework": 3244, + "Ġmedicine": 3245, + "azz": 3246, + "met": 3247, + "Ġplayful": 3248, + "Ġlocked": 3249, + "Ġgoat": 3250, + "Ġaccept": 3251, + "Ġcompass": 3252, + "Ġsurf": 3253, + "Ġcompetit": 3254, + "ext": 3255, + "kn": 3256, + "Ġsy": 3257, + "Ġsil": 3258, + "Ġbeet": 3259, + "ĠMar": 3260, + "uring": 3261, + "Ġchir": 3262, + "ĠEllie": 3263, + "Ġballs": 3264, + "ÅĵLet": 3265, + "Ġbutterflies": 3266, + "Ġguard": 3267, + "Ġcrayon": 3268, + "Ġdiscover": 3269, + "Ġparade": 3270, + "Ġmixed": 3271, + "Ġvalu": 3272, + "Ġhelmet": 3273, + "ĠBrownie": 3274, + "Em": 3275, + "ject": 3276, + "squ": 3277, + "ud": 3278, + "Ġgum": 3279, + "ature": 3280, + "Ġisland": 3281, + "Ġcost": 3282, + "Ġmole": 3283, + "Ġnicely": 3284, + "Ġkinds": 3285, + "Ġraced": 3286, + "Ġstove": 3287, + "Ġlaughs": 3288, + "Ġmarch": 3289, + "Ġfalls": 3290, + "Ġpersist": 3291, + "ĠBaby": 3292, + "ophie": 3293, + "Alice": 3294, + "ience": 3295, + "ito": 3296, + "Ġcage": 3297, + "Ġgoose": 3298, + "Ġcoins": 3299, + "Ġtrump": 3300, + "Ġmosqu": 3301, + "Ġserious": 3302, + "Ġmosquito": 3303, + "Ġflex": 3304, + "Ġcro": 3305, + "elon": 3306, + "Ġexcla": 3307, + "Ġchoose": 3308, + "Ġexplored": 3309, + "Ġunl": 3310, + "Ġbrothers": 3311, + "Ġguil": 3312, + "Ġnumbers": 3313, + "Ġtrumpet": 3314, + "Ġicy": 3315, + "Ġmoms": 3316, + "iew": 3317, + "Ġview": 3318, + "Ġenorm": 3319, + "Ġfireman": 3320, + "ĠMrs": 3321, + "Ġpractice": 3322, + "Ġenormous": 3323, + "wards": 3324, + "Ġsup": 3325, + "Ġvolc": 3326, + "Ġmeans": 3327, + "Ġpointing": 3328, + "Ġvaluable": 3329, + "Ġflexible": 3330, + "Ġexclaimed": 3331, + "Ġguilty": 3332, + "yal": 3333, + "Ġcam": 3334, + "Ġpipe": 3335, + "ision": 3336, + "ĠMy": 3337, + "Ġuseful": 3338, + "Ġfavour": 3339, + "Ġcrazy": 3340, + "Ġphot": 3341, + "known": 3342, + "Ġvolcano": 3343, + "Ġwake": 3344, + "Ġcity": 3345, + "omed": 3346, + "llip": 3347, + "Ġforth": 3348, + "Ġlollip": 3349, + "Ġseal": 3350, + "Ġallowed": 3351, + "Ġscare": 3352, + "Ġbrightly": 3353, + "Ġmarble": 3354, + "Ġpopular": 3355, + "Ġflashlight": 3356, + "Ġcompassion": 3357, + "Ġlollipop": 3358, + "ĠU": 3359, + "Ġing": 3360, + "Ġhunt": 3361, + "Ġdull": 3362, + "erry": 3363, + "Ġstat": 3364, + "Ġshel": 3365, + "!\".": 3366, + "Ġunknown": 3367, + "Ġhats": 3368, + "Ġshoot": 3369, + "redient": 3370, + "Ġbelonged": 3371, + "Ġfavourite": 3372, + "Ġingredient": 3373, + "Ġpil": 3374, + "Ġloyal": 3375, + "Ġbrace": 3376, + "ĠWhenever": 3377, + "Ġdreams": 3378, + "Ġkicked": 3379, + "Ġseeds": 3380, + "Ġtwirled": 3381, + "Ġsuper": 3382, + "time": 3383, + "Ġfine": 3384, + "Ġbees": 3385, + "Ġsofa": 3386, + "Ġshark": 3387, + "ĠMike": 3388, + "Ġanx": 3389, + "Ġsadly": 3390, + "Ġshowing": 3391, + "Ġways": 3392, + "Ġtrucks": 3393, + "Ġdelicate": 3394, + "ations": 3395, + "Ġrepe": 3396, + "Ġimpressive": 3397, + "Ġpoured": 3398, + "Ġpirate": 3399, + "Ġwhispered": 3400, + "Ġharmless": 3401, + "aff": 3402, + "Ġturt": 3403, + "Ġtied": 3404, + "aroo": 3405, + "ado": 3406, + "arth": 3407, + "Ġgrace": 3408, + "Ġhandle": 3409, + "Ġenc": 3410, + "Ġdeliver": 3411, + "Benny": 3412, + "Ġrushed": 3413, + "Ġusing": 3414, + "angaroo": 3415, + "Ġshape": 3416, + "Ġbushes": 3417, + "Ġpersistent": 3418, + "Ġgor": 3419, + "ĠSp": 3420, + "Ġkangaroo": 3421, + "Ġang": 3422, + "Ġoffice": 3423, + "Ġanxious": 3424, + "Go": 3425, + "alous": 3426, + "Ġspell": 3427, + "ĠWhile": 3428, + "iable": 3429, + "Ġpotato": 3430, + "Ġjealous": 3431, + "Ġwrapped": 3432, + "Ġfour": 3433, + "Ġcurt": 3434, + "Ġlog": 3435, + "Ġcoal": 3436, + "Ġreliable": 3437, + "Ġcurtain": 3438, + "Daisy": 3439, + "Ġsudden": 3440, + "mbol": 3441, + "ÅĵYes": 3442, + "aches": 3443, + "ĠPe": 3444, + "Ġpainted": 3445, + "ĠToby": 3446, + "Ġcere": 3447, + "gu": 3448, + "Ġtummy": 3449, + "Ġthose": 3450, + "ouses": 3451, + "uddy": 3452, + "Ġcush": 3453, + "Ġmetal": 3454, + "Ġsymbol": 3455, + "Ġbeetle": 3456, + "Ġcamera": 3457, + "Ġhouses": 3458, + "ilt": 3459, + "Ġcelebrate": 3460, + "Ġdelighted": 3461, + "Ġgorilla": 3462, + "Ġtooth": 3463, + "Ġthirst": 3464, + "keep": 3465, + "Ġshiver": 3466, + "Ġreward": 3467, + "Ġspace": 3468, + "Ġtravel": 3469, + "Ġrubbed": 3470, + "Ġprint": 3471, + "Ġdriving": 3472, + "Ġmarry": 3473, + "Ġwarned": 3474, + "Ġsoldier": 3475, + "Ġmotorcy": 3476, + "cut": 3477, + "Ġson": 3478, + "Ġtor": 3479, + "arlie": 3480, + "Ġhero": 3481, + "ief": 3482, + "Ġcarp": 3483, + "Ġexcitedly": 3484, + "Ġtemp": 3485, + "ĠPete": 3486, + "ĠBobo": 3487, + "cod": 3488, + "Ġhung": 3489, + "Ġgrowing": 3490, + "upid": 3491, + "Ġthinking": 3492, + "Ġtunn": 3493, + "While": 3494, + "cu": 3495, + "ipp": 3496, + "Ġhay": 3497, + "Ġyet": 3498, + "Ġvine": 3499, + "Ġacorn": 3500, + "icycle": 3501, + "Ġprep": 3502, + "Ġreminded": 3503, + "Ġgasped": 3504, + "Ġflute": 3505, + "Ġordinary": 3506, + "Ġcrocod": 3507, + "Ġamb": 3508, + "Ġcur": 3509, + "ĠBet": 3510, + "ĠMandy": 3511, + "Ġpup": 3512, + "Ġcreatures": 3513, + "Ġhurry": 3514, + "Ġspoiled": 3515, + "Ġtimes": 3516, + "Ġmis": 3517, + "Ġrice": 3518, + "Ġstupid": 3519, + "Ġshine": 3520, + "Ġresist": 3521, + "ugged": 3522, + "Ġlaw": 3523, + "Ġskull": 3524, + "coa": 3525, + "Ġmissing": 3526, + "Ġwheat": 3527, + "Ġsupport": 3528, + "Ġingredients": 3529, + "aid": 3530, + "loo": 3531, + "Ġcase": 3532, + "ery": 3533, + "Ġwashed": 3534, + "Ġdove": 3535, + "Ġspilled": 3536, + "Ġscoot": 3537, + "Ġmeal": 3538, + "Ġmule": 3539, + "ĠDucky": 3540, + "Ġmoder": 3541, + "arsh": 3542, + "ÅĵWhat": 3543, + "Ġtrick": 3544, + "Ġsett": 3545, + "ĠTweetie": 3546, + "Ġbounce": 3547, + "Ġfilthy": 3548, + "Ġbase": 3549, + "Ġhoop": 3550, + "Ġfre": 3551, + "Ġstumbled": 3552, + "Ġsoar": 3553, + "Ġweal": 3554, + "Ġseem": 3555, + "Ġword": 3556, + "Ġlab": 3557, + "ĠDave": 3558, + "ĠFr": 3559, + "Ġbadly": 3560, + "Ġhairy": 3561, + "Ġcrawl": 3562, + "Ġlively": 3563, + "Ġsteps": 3564, + "ÅĵLetâ": 3565, + "Ġbracelet": 3566, + "Ġcereal": 3567, + "Ġthirsty": 3568, + "Ġcarpet": 3569, + "Ġwool": 3570, + "ither": 3571, + "Ġgoal": 3572, + "Ġbackpack": 3573, + "Ġtruth": 3574, + "Ġcompl": 3575, + "Ġcheap": 3576, + "Ġdisg": 3577, + "Ġplaced": 3578, + "Ġearly": 3579, + "Ġdecorate": 3580, + "Ġnuts": 3581, + "Ow": 3582, + "cream": 3583, + "Ġbump": 3584, + "Ġthemselves": 3585, + "ices": 3586, + "Ġhelpless": 3587, + "Ġclum": 3588, + "Ġscatter": 3589, + "Ġscale": 3590, + "Ġnews": 3591, + "usting": 3592, + "Ġenv": 3593, + "âĤ¬âĢ": 3594, + "Ġtelling": 3595, + "Ġswinging": 3596, + "Ġsparkled": 3597, + "Ġcarrying": 3598, + "gging": 3599, + "ines": 3600, + "ric": 3601, + "Ġfriendship": 3602, + "Ġcareless": 3603, + "Ġsne": 3604, + "Ġdoesn": 3605, + "Ġfalling": 3606, + "Ġcomput": 3607, + "Ġpale": 3608, + "Ġpeeked": 3609, + "Ġcompassionate": 3610, + "com": 3611, + "vice": 3612, + "Ġsend": 3613, + "Ġstage": 3614, + "Ġmem": 3615, + "Ġchalk": 3616, + "Ġchasing": 3617, + "Ġpebble": 3618, + "ĠEverything": 3619, + "Ġcube": 3620, + "Ġpige": 3621, + "Ġcourage": 3622, + "Ġfingers": 3623, + "Ġturtle": 3624, + "bled": 3625, + "Ġsink": 3626, + "Ġbicycle": 3627, + "Ġshaking": 3628, + "Ġsleeping": 3629, + "Ġneighbour": 3630, + "Ġmodern": 3631, + "Ġdisgusting": 3632, + "Ġclumsy": 3633, + "gest": 3634, + "Ġhook": 3635, + "Ġcher": 3636, + "Ġnail": 3637, + "Ġbiggest": 3638, + "iches": 3639, + "Ġbul": 3640, + "Ġending": 3641, + "Ġdead": 3642, + "Ġtriang": 3643, + "ulance": 3644, + "Ġspoke": 3645, + "Ġpracticed": 3646, + "Ġambulance": 3647, + "Ġcomputer": 3648, + "ibb": 3649, + "Ġlamp": 3650, + "Ġstation": 3651, + "Ġchar": 3652, + "Ġwallet": 3653, + "ĠToday": 3654, + "Ġpackage": 3655, + "Ġmicrop": 3656, + "Ġcushion": 3657, + "keeper": 3658, + "Ġcrocodile": 3659, + "Ġmicrophone": 3660, + "under": 3661, + "ĠAlex": 3662, + "Ġthoughtful": 3663, + "Ġjolly": 3664, + "Ġpengu": 3665, + "Ġpumpkin": 3666, + "Ġcostum": 3667, + "Ġfive": 3668, + "Ġdough": 3669, + "Ġpony": 3670, + "Ġol": 3671, + "imi": 3672, + "Ġstairs": 3673, + "Ġknock": 3674, + "Ġgrand": 3675, + "Ġintell": 3676, + "Ġuniver": 3677, + "Ġbrighter": 3678, + "Ġopens": 3679, + "Ġdreaming": 3680, + "Ġbounced": 3681, + "Ġquestion": 3682, + "Ġstatue": 3683, + "Ġwine": 3684, + "isk": 3685, + "ĠAlice": 3686, + "Ġcocoa": 3687, + "ĠYour": 3688, + "Ġpepper": 3689, + "Ġbeauty": 3690, + "Ġperm": 3691, + "Ġpainting": 3692, + "Ġshoe": 3693, + "Ġelev": 3694, + "Everyone": 3695, + "hino": 3696, + "Ġwealthy": 3697, + "Ġintellig": 3698, + "noon": 3699, + "Ġsuit": 3700, + "Ġmild": 3701, + "Ġnature": 3702, + "ans": 3703, + "Ġhappier": 3704, + "Ġneat": 3705, + "Ġstarts": 3706, + "ucked": 3707, + "Ġafternoon": 3708, + "Ġtorn": 3709, + "Ġelevator": 3710, + "gen": 3711, + "ti": 3712, + "Ġten": 3713, + "Ġharsh": 3714, + "Ġped": 3715, + "atient": 3716, + "orant": 3717, + "Ġrece": 3718, + "Ġjet": 3719, + "illie": 3720, + "ĠRed": 3721, + "Ġreading": 3722, + "Ġsearching": 3723, + "Ġbasketball": 3724, + "Ġbarber": 3725, + "Ġspeed": 3726, + "See": 3727, + "awn": 3728, + "Ġstack": 3729, + "ĠMummy": 3730, + "Ġclock": 3731, + "ĠGive": 3732, + "Ġmagn": 3733, + "Ġleaving": 3734, + "Ġcupboard": 3735, + "Ġdesign": 3736, + "Ġslides": 3737, + "Ġsandwiches": 3738, + "Ġcandle": 3739, + "aulif": 3740, + "Ġriding": 3741, + "Ġfresh": 3742, + "oppy": 3743, + "Ġshocked": 3744, + "Ġener": 3745, + "Ġjumps": 3746, + "Ġhaircut": 3747, + "Ġrespectful": 3748, + "Ġcooking": 3749, + "Ġticket": 3750, + "Ġencou": 3751, + "Ġtunnel": 3752, + "auliflower": 3753, + "Give": 3754, + "Ġsle": 3755, + "reen": 3756, + "Ġmay": 3757, + "Ġthief": 3758, + "Ġyawn": 3759, + "Ġfort": 3760, + "Ġalert": 3761, + "Ġsorts": 3762, + "Ġimpatient": 3763, + "Ġcreate": 3764, + "ailable": 3765, + "Ġwheels": 3766, + "Ġvalue": 3767, + "Ġprize": 3768, + "Mary": 3769, + "ye": 3770, + "het": 3771, + "Ġsent": 3772, + "Ġbent": 3773, + "ince": 3774, + "Ġcomet": 3775, + "Ġcamp": 3776, + "Ġcauliflower": 3777, + "Ġmelon": 3778, + "Ġgem": 3779, + "Ġitself": 3780, + "iron": 3781, + "Ġmush": 3782, + "Ġmonkeys": 3783, + "Ġsalad": 3784, + "Ġdestro": 3785, + "Ġrescue": 3786, + "Ġscooter": 3787, + "Ġintelligent": 3788, + "az": 3789, + "Ġdug": 3790, + "iling": 3791, + "anger": 3792, + "Ġroof": 3793, + "Ġrhino": 3794, + "Ġscold": 3795, + "aghet": 3796, + "Ġexper": 3797, + "Ġbarking": 3798, + "Ġbananas": 3799, + "Ġignorant": 3800, + "Ġmicro": 3801, + "Ġavailable": 3802, + "Ġgraceful": 3803, + "Ġmotorcycle": 3804, + "aghetti": 3805, + "hood": 3806, + "ĠSn": 3807, + "Ġearth": 3808, + "Ġboots": 3809, + "Ġspaghetti": 3810, + "ummer": 3811, + "Ġagree": 3812, + "Ġbull": 3813, + "Ġanywhere": 3814, + "ĠGeorge": 3815, + "Ġbehave": 3816, + "ĠKim": 3817, + "Ġpaid": 3818, + "scope": 3819, + "Ġstretched": 3820, + "Ġfearful": 3821, + "Ġavo": 3822, + "Ġmicroscope": 3823, + "If": 3824, + "io": 3825, + "Ġnoteb": 3826, + "ĠEventually": 3827, + "Ġolder": 3828, + "Ġsniff": 3829, + "Ġadvice": 3830, + "Ġstops": 3831, + "Ġperform": 3832, + "Ġfurther": 3833, + "Ġenvious": 3834, + "Ġpigeon": 3835, + "Ġmushroom": 3836, + "Ġnotebook": 3837, + "ubby": 3838, + "Ġpun": 3839, + "ats": 3840, + "Ġnurse": 3841, + "orable": 3842, + "Ġthunder": 3843, + "Ġclapping": 3844, + "Ġvide": 3845, + "Ġbuilt": 3846, + "ĠCl": 3847, + "Ġuncom": 3848, + "Ġshouting": 3849, + "Ġarrow": 3850, + "Ġdrawer": 3851, + "Ġprov": 3852, + "fortable": 3853, + "Ġtomato": 3854, + "Ġmeeting": 3855, + "Ġobject": 3856, + "Ġyogurt": 3857, + "Ġtemple": 3858, + "Ġuncomfortable": 3859, + "ef": 3860, + "ues": 3861, + "ĠTony": 3862, + "Ġloop": 3863, + "Ġchubby": 3864, + "Ġswitch": 3865, + "Ġskip": 3866, + "Ġadorable": 3867, + "ÅĵIt": 3868, + "Ġcounting": 3869, + "Ġcooked": 3870, + "Ġpist": 3871, + "arian": 3872, + "enry": 3873, + "Ġstared": 3874, + "Ġshut": 3875, + "icop": 3876, + "Ġquite": 3877, + "Ġsnuggled": 3878, + "ĠGrandpa": 3879, + "Ġtroubled": 3880, + "Ġgiggle": 3881, + "Ġblowing": 3882, + "Ġhelicop": 3883, + "Ġfisher": 3884, + "Ġpistol": 3885, + "Ġhelicopter": 3886, + "par": 3887, + "ĠSophie": 3888, + "ooped": 3889, + "ips": 3890, + "Ġnowhere": 3891, + "Ġforgave": 3892, + "Ġmarched": 3893, + "Ġtreasures": 3894, + "host": 3895, + "Ġpassport": 3896, + "Ġstirred": 3897, + "Ġpushing": 3898, + "Ġcompetitive": 3899, + "Ġdestroy": 3900, + "Yay": 3901, + "aked": 3902, + "oe": 3903, + "tter": 3904, + "Ġgir": 3905, + "Ġghost": 3906, + "anic": 3907, + "kelet": 3908, + "htub": 3909, + "Ġreef": 3910, + "Ġsepar": 3911, + "opard": 3912, + "play": 3913, + "Ġenvel": 3914, + "Ġbreat": 3915, + "Ġrepair": 3916, + "Ġfootball": 3917, + "Ġbathtub": 3918, + "Ġrubber": 3919, + "Ġangel": 3920, + "Ġtriangle": 3921, + "Ġled": 3922, + "Ġmill": 3923, + "ĠSh": 3924, + "chanic": 3925, + "Ġnames": 3926, + "Ġleopard": 3927, + "Ġnobody": 3928, + "Ġanyway": 3929, + "uct": 3930, + "ctus": 3931, + "Ġshouldn": 3932, + "Ġzeb": 3933, + "Ġgiven": 3934, + "Ġholds": 3935, + "Ġbarrel": 3936, + "Ġbandage": 3937, + "Ġenth": 3938, + "Daddy": 3939, + "Ġzebra": 3940, + "ives": 3941, + "ube": 3942, + "year": 3943, + "Ġpor": 3944, + "Ġpand": 3945, + "Ġotter": 3946, + "Ġribb": 3947, + "Ġshield": 3948, + "Ġmechanic": 3949, + "Ġdonâ": 3950, + "Ġdisag": 3951, + "Ġdisplay": 3952, + "Ġmusician": 3953, + "ceros": 3954, + "Ġspeak": 3955, + "Ġcab": 3956, + "Ġcactus": 3957, + "ĠBetsy": 3958, + "Ġbulb": 3959, + "Ch": 3960, + "eath": 3961, + "ikes": 3962, + "xy": 3963, + "Ġtube": 3964, + "Ġahead": 3965, + "Ġsummer": 3966, + "Ġskelet": 3967, + "Ġpurse": 3968, + "Ġknob": 3969, + "Ġdidnâ": 3970, + "Ġcouldnâ": 3971, + "Ġcolours": 3972, + "Ġwindows": 3973, + "oomy": 3974, + "Ġgifted": 3975, + "Ġcartoon": 3976, + "Ġrhinoceros": 3977, + "gl": 3978, + "heart": 3979, + "Ġsize": 3980, + "Ġbet": 3981, + "ĠMiss": 3982, + "Ġscrew": 3983, + "Ġflour": 3984, + "Ġblin": 3985, + "apa": 3986, + "Ġdishes": 3987, + "Ġraining": 3988, + "Ġhoping": 3989, + "Ġshovel": 3990, + "Ġmint": 3991, + "Ġjellyfish": 3992, + "Ġsweater": 3993, + "Ġcharming": 3994, + "Ġpenguin": 3995, + "Ġenvelope": 3996, + "Ġenthus": 3997, + "Ġcabin": 3998, + "rec": 3999, + "Ġpant": 4000, + "Ġthread": 4001, + "Ġinse": 4002, + "Ġstaff": 4003, + "Ġliz": 4004, + "Ġunpack": 4005, + "Ġwriting": 4006, + "Ġtrash": 4007, + "Ġrolling": 4008, + "Ġwaffle": 4009, + "Ġiron": 4010, + "Ġwing": 4011, + "erable": 4012, + "Ġsheet": 4013, + "ĠBuddy": 4014, + "Ġshadow": 4015, + "Ġroared": 4016, + "Ġmuff": 4017, + "Ġairport": 4018, + "Ġceiling": 4019, + "Ġattic": 4020, + "Ġorganize": 4021, + "Ġstretch": 4022, + "Ġgrapes": 4023, + "Ġobs": 4024, + "ĠAre": 4025, + "ĠWould": 4026, + "ĠInst": 4027, + "Ġpiano": 4028, + "Ġsalt": 4029, + "Ġmiserable": 4030, + "Ġcord": 4031, + "Ġdess": 4032, + "Ġrefused": 4033, + "Ġscreen": 4034, + "ĠDan": 4035, + "Ġstrugg": 4036, + "Ġneedle": 4037, + "Ġbears": 4038, + "ĠAndy": 4039, + "Ġheaded": 4040, + "Ġsticky": 4041, + "Ġfreez": 4042, + "Ġzoomed": 4043, + "Ġimagined": 4044, + "Ġmedal": 4045, + "Ġmagnet": 4046, + "reci": 4047, + "Ġobser": 4048, + "Ġer": 4049, + "Ġlives": 4050, + "ĠAl": 4051, + "ĠPat": 4052, + "Ġappreci": 4053, + "asses": 4054, + "Ġsailor": 4055, + "Ġminute": 4056, + "Ġfisherman": 4057, + "vision": 4058, + "Ġrat": 4059, + "aled": 4060, + "Ġbrus": 4061, + "Ġchance": 4062, + "Ġpost": 4063, + "Ġsungl": 4064, + "Ġcheek": 4065, + "reeze": 4066, + "Ġdeaf": 4067, + "Ġrules": 4068, + "Ġfrogs": 4069, + "Ġairpl": 4070, + "ĠGod": 4071, + "Ġstrawberry": 4072, + "Ġslept": 4073, + "Ġlizard": 4074, + "Ġsunglasses": 4075, + "Ġluck": 4076, + "Ġinf": 4077, + "Ġbeep": 4078, + "ĠBunny": 4079, + "Ġmeas": 4080, + "Ġrod": 4081, + "Ġorder": 4082, + "Ġrestless": 4083, + "Ġsandbox": 4084, + "Ġmessage": 4085, + "ĠJoey": 4086, + "Ġcomplet": 4087, + "Ġattract": 4088, + "Ġgolden": 4089, + "Grandma": 4090, + "Ġpilot": 4091, + "Ġporch": 4092, + "Little": 4093, + "ler": 4094, + "ses": 4095 + }, + "merges": [ + [ + "h", + "e" + ], + [ + "Ġ", + "t" + ], + [ + "Ġ", + "a" + ], + [ + "Ġ", + "s" + ], + [ + "Ġ", + "w" + ], + [ + "n", + "d" + ], + [ + "Ġt", + "he" + ], + [ + "e", + "d" + ], + [ + "Ġa", + "nd" + ], + [ + "Ġt", + "o" + ], + [ + "Ġ", + "b" + ], + [ + "i", + "n" + ], + [ + "Ġ", + "h" + ], + [ + "Ġw", + "a" + ], + [ + "r", + "e" + ], + [ + "Ġ", + "f" + ], + [ + "i", + "t" + ], + [ + "o", + "u" + ], + [ + "Ġ", + "c" + ], + [ + "Ġ", + "l" + ], + [ + "Ġ", + "he" + ], + [ + "Ġ", + "d" + ], + [ + "e", + "r" + ], + [ + "Ġwa", + "s" + ], + [ + "Ġ", + "m" + ], + [ + "Ġ", + "p" + ], + [ + "o", + "m" + ], + [ + "Ġ", + "T" + ], + [ + "Ġ", + "o" + ], + [ + "a", + "y" + ], + [ + "a", + "r" + ], + [ + "in", + "g" + ], + [ + "i", + "s" + ], + [ + "Ġ", + "g" + ], + [ + "i", + "l" + ], + [ + "i", + "d" + ], + [ + "a", + "t" + ], + [ + "e", + "n" + ], + [ + "Ġ", + "n" + ], + [ + "Ġs", + "a" + ], + [ + "Ġh", + "a" + ], + [ + "Ġ", + "S" + ], + [ + "i", + "m" + ], + [ + "a", + "n" + ], + [ + "ĠT", + "he" + ], + [ + "o", + "r" + ], + [ + "o", + "n" + ], + [ + "Ġ", + "it" + ], + [ + "Ġt", + "h" + ], + [ + "l", + "l" + ], + [ + "l", + "e" + ], + [ + "Ġ", + "H" + ], + [ + "Ġhe", + "r" + ], + [ + "e", + "t" + ], + [ + "o", + "t" + ], + [ + "i", + "r" + ], + [ + "ĠS", + "he" + ], + [ + "ĠH", + "e" + ], + [ + "v", + "er" + ], + [ + "e", + "s" + ], + [ + "Ġ", + "in" + ], + [ + "u", + "t" + ], + [ + "o", + "w" + ], + [ + "c", + "k" + ], + [ + "Ġ", + "e" + ], + [ + "Ġ", + "u" + ], + [ + "l", + "d" + ], + [ + "ĠThe", + "y" + ], + [ + "o", + "o" + ], + [ + "i", + "g" + ], + [ + "Ġsa", + "id" + ], + [ + "a", + "m" + ], + [ + "il", + "y" + ], + [ + "Ġb", + "e" + ], + [ + "Ġ", + "y" + ], + [ + "Ġ", + "r" + ], + [ + "Ġs", + "t" + ], + [ + "c", + "e" + ], + [ + "Ġs", + "he" + ], + [ + "Ġ", + "\"" + ], + [ + "p", + "p" + ], + [ + "k", + "e" + ], + [ + "it", + "h" + ], + [ + "O", + "n" + ], + [ + "Ġ", + "I" + ], + [ + "Ġw", + "ith" + ], + [ + "v", + "e" + ], + [ + "L", + "ily" + ], + [ + "Ġo", + "n" + ], + [ + "Ġo", + "f" + ], + [ + "Ġs", + "o" + ], + [ + "Ġh", + "is" + ], + [ + "k", + "ed" + ], + [ + "r", + "i" + ], + [ + "n", + "t" + ], + [ + "ver", + "y" + ], + [ + "Ġp", + "l" + ], + [ + "Ġd", + "ay" + ], + [ + "a", + "d" + ], + [ + "Ġy", + "ou" + ], + [ + "Ġth", + "at" + ], + [ + "Ġu", + "p" + ], + [ + "Ġha", + "d" + ], + [ + "s", + "t" + ], + [ + "Ġpl", + "ay" + ], + [ + "Ġthe", + "y" + ], + [ + "Ġ", + "Lily" + ], + [ + "Ġw", + "e" + ], + [ + "Ġm", + "om" + ], + [ + "m", + "y" + ], + [ + "Ġf", + "or" + ], + [ + "e", + "l" + ], + [ + "ou", + "ld" + ], + [ + "u", + "n" + ], + [ + "Ġ", + "B" + ], + [ + "'", + "s" + ], + [ + "it", + "t" + ], + [ + "en", + "t" + ], + [ + "Ġha", + "pp" + ], + [ + "T", + "he" + ], + [ + "c", + "h" + ], + [ + "Ġl", + "i" + ], + [ + "ou", + "t" + ], + [ + "Ġwa", + "nt" + ], + [ + "Ġs", + "h" + ], + [ + "he", + "r" + ], + [ + "l", + "y" + ], + [ + "im", + "e" + ], + [ + "itt", + "le" + ], + [ + "ou", + "nd" + ], + [ + "Ġ", + "very" + ], + [ + "Ġt", + "ime" + ], + [ + "om", + "e" + ], + [ + "Ġl", + "ittle" + ], + [ + "Ġthe", + "re" + ], + [ + "s", + "e" + ], + [ + "Ġd", + "o" + ], + [ + "Ġw", + "h" + ], + [ + "a", + "ll" + ], + [ + "Ġ", + "k" + ], + [ + "e", + "nd" + ], + [ + "a", + "l" + ], + [ + "h", + "t" + ], + [ + "Ġn", + "e" + ], + [ + "Ġ", + "re" + ], + [ + "Ġn", + "ot" + ], + [ + "Ġhapp", + "y" + ], + [ + "Ġ", + "Ċ" + ], + [ + "Ġb", + "ig" + ], + [ + "Ġ", + "M" + ], + [ + "Ġb", + "ut" + ], + [ + "Ġs", + "m" + ], + [ + "a", + "ck" + ], + [ + "Ġsa", + "w" + ], + [ + "ĠI", + "t" + ], + [ + "Ġa", + "s" + ], + [ + "Ġa", + "n" + ], + [ + "r", + "a" + ], + [ + "ri", + "end" + ], + [ + "Ġf", + "riend" + ], + [ + "id", + "e" + ], + [ + "On", + "e" + ], + [ + "r", + "y" + ], + [ + "'", + "t" + ], + [ + "v", + "ed" + ], + [ + "Ġ", + "is" + ], + [ + "On", + "ce" + ], + [ + "a", + "ke" + ], + [ + ".", + "\"" + ], + [ + "Ġwe", + "re" + ], + [ + "t", + "er" + ], + [ + "u", + "g" + ], + [ + "Ġl", + "oo" + ], + [ + "Ġl", + "o" + ], + [ + "o", + "re" + ], + [ + "e", + "c" + ], + [ + "ĠT", + "im" + ], + [ + "Ġh", + "im" + ], + [ + "Ġb", + "o" + ], + [ + "!", + "\"" + ], + [ + "Ġto", + "o" + ], + [ + "Ġg", + "o" + ], + [ + "Ġup", + "on" + ], + [ + "ir", + "l" + ], + [ + "Ġ", + "j" + ], + [ + "Ġwant", + "ed" + ], + [ + "Ġg", + "irl" + ], + [ + "Ġs", + "e" + ], + [ + "Ġ", + "out" + ], + [ + "ar", + "d" + ], + [ + "w", + "ay" + ], + [ + "Ġs", + "p" + ], + [ + "il", + "l" + ], + [ + "i", + "nd" + ], + [ + "Ġthe", + "m" + ], + [ + "Ġc", + "ould" + ], + [ + "f", + "u" + ], + [ + "he", + "n" + ], + [ + "Ġa", + "t" + ], + [ + "u", + "r" + ], + [ + "Ġd", + "id" + ], + [ + "Ġsm", + "il" + ], + [ + "Ġthe", + "ir" + ], + [ + "Ġa", + "re" + ], + [ + "Ġe", + "x" + ], + [ + "Ġ", + "A" + ], + [ + "a", + "in" + ], + [ + "Ġw", + "ent" + ], + [ + "ar", + "t" + ], + [ + "he", + "d" + ], + [ + "r", + "om" + ], + [ + "i", + "c" + ], + [ + "r", + "ound" + ], + [ + "Ġha", + "ve" + ], + [ + "Ġn", + "am" + ], + [ + "l", + "p" + ], + [ + "Ġa", + "ll" + ], + [ + "Ġ", + "J" + ], + [ + "fu", + "l" + ], + [ + "Ġk", + "n" + ], + [ + "h", + "ing" + ], + [ + "oo", + "d" + ], + [ + "Ġhe", + "lp" + ], + [ + "ig", + "ht" + ], + [ + "Ġfriend", + "s" + ], + [ + "on", + "e" + ], + [ + "ar", + "k" + ], + [ + "Ġb", + "ack" + ], + [ + "u", + "m" + ], + [ + "Ġc", + "an" + ], + [ + "Ġnam", + "ed" + ], + [ + "Ġc", + "l" + ], + [ + "?", + "\"" + ], + [ + "Ġf", + "un" + ], + [ + "a", + "re" + ], + [ + "ĠB", + "en" + ], + [ + "Ġlo", + "ved" + ], + [ + "Ġa", + "l" + ], + [ + "el", + "t" + ], + [ + "ĠTim", + "my" + ], + [ + "Ġ", + "One" + ], + [ + "o", + "p" + ], + [ + "s", + "ide" + ], + [ + "Ġl", + "e" + ], + [ + "Ġn", + "o" + ], + [ + "Ġs", + "c" + ], + [ + "ĠT", + "om" + ], + [ + "Ġf", + "elt" + ], + [ + "ou", + "g" + ], + [ + "Ġsmil", + "ed" + ], + [ + "i", + "ck" + ], + [ + "Ġas", + "ked" + ], + [ + "Y", + "ou" + ], + [ + "Ġto", + "y" + ], + [ + "Ġm", + "an" + ], + [ + "Ġa", + "round" + ], + [ + "am", + "e" + ], + [ + "Ġf", + "e" + ], + [ + "Ġs", + "ay" + ], + [ + "Ġbo", + "y" + ], + [ + "Ġs", + "ome" + ], + [ + "Ġloo", + "ked" + ], + [ + "u", + "re" + ], + [ + "om", + "et" + ], + [ + "Ġb", + "r" + ], + [ + "Ġw", + "ould" + ], + [ + "Ġm", + "e" + ], + [ + "Ġb", + "ir" + ], + [ + "Ġli", + "ke" + ], + [ + "g", + "et" + ], + [ + "Ġst", + "art" + ], + [ + "Ġr", + "o" + ], + [ + "a", + "s" + ], + [ + "Ġse", + "e" + ], + [ + "Ġ", + "W" + ], + [ + "i", + "ce" + ], + [ + "on", + "g" + ], + [ + "Ġbir", + "d" + ], + [ + "Ġs", + "omet" + ], + [ + "d", + "d" + ], + [ + "Ġw", + "or" + ], + [ + "ad", + "e" + ], + [ + "i", + "e" + ], + [ + "k", + "ing" + ], + [ + "Ġa", + "g" + ], + [ + "ow", + "n" + ], + [ + "Ġt", + "re" + ], + [ + "Ġf", + "a" + ], + [ + "Ġa", + "way" + ], + [ + "Ġwh", + "at" + ], + [ + "ing", + "s" + ], + [ + "Ġstart", + "ed" + ], + [ + "get", + "her" + ], + [ + "Ġr", + "an" + ], + [ + "Ã", + "¢" + ], + [ + "â", + "Ĥ" + ], + [ + "âĤ", + "¬" + ], + [ + "ar", + "ed" + ], + [ + "Ġm", + "ake" + ], + [ + "ĠB", + "ut" + ], + [ + "it", + "ed" + ], + [ + "i", + "f" + ], + [ + "ou", + "d" + ], + [ + "Ġm", + "ade" + ], + [ + "Ġto", + "gether" + ], + [ + "Ġsomet", + "hing" + ], + [ + "Ġex", + "c" + ], + [ + "a", + "g" + ], + [ + "Ġc", + "o" + ], + [ + "Ġp", + "ark" + ], + [ + "Ġne", + "w" + ], + [ + "Ġsa", + "d" + ], + [ + "Ġp", + "ut" + ], + [ + ",", + "\"" + ], + [ + "Ġf", + "rom" + ], + [ + "b", + "le" + ], + [ + "t", + "her" + ], + [ + "Ġp", + "r" + ], + [ + "Ġm", + "u" + ], + [ + "Ġc", + "ar" + ], + [ + "Ġh", + "ome" + ], + [ + "Ġ", + "You" + ], + [ + "Ġthe", + "n" + ], + [ + "Ġw", + "hen" + ], + [ + "Ġf", + "ound" + ], + [ + "e", + "ll" + ], + [ + "Ġo", + "ther" + ], + [ + "Ġag", + "ain" + ], + [ + "Ġc", + "h" + ], + [ + "Ġd", + "ec" + ], + [ + "Ġwh", + "o" + ], + [ + "Ġl", + "a" + ], + [ + "ri", + "ed" + ], + [ + "s", + "s" + ], + [ + "Ġg", + "ood" + ], + [ + "Ġh", + "ug" + ], + [ + "Ġ", + "L" + ], + [ + "pp", + "ed" + ], + [ + "Ġwa", + "l" + ], + [ + "e", + "p" + ], + [ + "all", + "y" + ], + [ + "Ġsay", + "s" + ], + [ + "Ġf", + "l" + ], + [ + "es", + "t" + ], + [ + "a", + "ch" + ], + [ + "Ġ", + "E" + ], + [ + "Ġexc", + "ited" + ], + [ + "p", + "l" + ], + [ + "q", + "u" + ], + [ + "oo", + "k" + ], + [ + "Ġg", + "et" + ], + [ + "oug", + "ht" + ], + [ + "Ġplay", + "ing" + ], + [ + "Ġg", + "ot" + ], + [ + "Ġs", + "w" + ], + [ + "ou", + "s" + ], + [ + "h", + "at" + ], + [ + "n", + "y" + ], + [ + "id", + "ed" + ], + [ + "u", + "ck" + ], + [ + "Ġth", + "ings" + ], + [ + "Ġe", + "very" + ], + [ + "Ġdec", + "ided" + ], + [ + "Ġc", + "ame" + ], + [ + "Ġbe", + "c" + ], + [ + "a", + "ve" + ], + [ + "r", + "o" + ], + [ + "a", + "x" + ], + [ + "Ġli", + "ked" + ], + [ + "Ġd", + "own" + ], + [ + "Ġdo", + "g" + ], + [ + "Ġsc", + "ared" + ], + [ + "Ġ", + "v" + ], + [ + "u", + "dd" + ], + [ + "u", + "st" + ], + [ + "Ġon", + "e" + ], + [ + "Ġf", + "ind" + ], + [ + "Ġb", + "l" + ], + [ + "Ġth", + "an" + ], + [ + "Ġ", + "D" + ], + [ + "ou", + "se" + ], + [ + "way", + "s" + ], + [ + "Ġkn", + "e" + ], + [ + "Ġdid", + "n" + ], + [ + "a", + "p" + ], + [ + "Ġmom", + "my" + ], + [ + "Ġc", + "are" + ], + [ + "Ġal", + "ways" + ], + [ + "Ġa", + "b" + ], + [ + "is", + "t" + ], + [ + "Ġd", + "ad" + ], + [ + "Ġfe", + "el" + ], + [ + "ar", + "a" + ], + [ + "b", + "b" + ], + [ + "ar", + "n" + ], + [ + "f", + "e" + ], + [ + "Ġyou", + "r" + ], + [ + "Ġout", + "side" + ], + [ + "u", + "e" + ], + [ + "Ġg", + "ra" + ], + [ + "an", + "t" + ], + [ + "Ġ", + "ke" + ], + [ + "ĠM", + "om" + ], + [ + "Ġtoo", + "k" + ], + [ + "Ġl", + "ot" + ], + [ + "n", + "n" + ], + [ + "Ġb", + "u" + ], + [ + "Ġab", + "out" + ], + [ + "es", + "s" + ], + [ + "Ġ", + "F" + ], + [ + "nd", + "er" + ], + [ + "Ġtre", + "e" + ], + [ + "ec", + "i" + ], + [ + "Ġloo", + "k" + ], + [ + "Ġp", + "o" + ], + [ + "Ġm", + "y" + ], + [ + "ou", + "r" + ], + [ + "Ġtoy", + "s" + ], + [ + "it", + "e" + ], + [ + "c", + "hed" + ], + [ + "Ġkne", + "w" + ], + [ + "Ġth", + "ought" + ], + [ + "en", + "ed" + ], + [ + "Ġle", + "arn" + ], + [ + "Ġin", + "t" + ], + [ + "Ġo", + "ld" + ], + [ + "Ġm", + "ore" + ], + [ + "nn", + "a" + ], + [ + "is", + "e" + ], + [ + "g", + "ed" + ], + [ + "Ġt", + "a" + ], + [ + "udd", + "en" + ], + [ + "eci", + "al" + ], + [ + "Ġsp", + "ecial" + ], + [ + "ĠM", + "ax" + ], + [ + "a", + "u" + ], + [ + "Ġw", + "ill" + ], + [ + "er", + "s" + ], + [ + "The", + "y" + ], + [ + "re", + "t" + ], + [ + "Ġp", + "e" + ], + [ + "Ġh", + "o" + ], + [ + "ĠS", + "am" + ], + [ + "Ġt", + "ake" + ], + [ + "Ġb", + "all" + ], + [ + "Ġkn", + "ow" + ], + [ + "Ġla", + "ug" + ], + [ + "f", + "ter" + ], + [ + "udden", + "ly" + ], + [ + "Ġc", + "at" + ], + [ + "Ġh", + "ow" + ], + [ + "i", + "ve" + ], + [ + "Ġt", + "r" + ], + [ + "Ġmu", + "ch" + ], + [ + "Ġan", + "y" + ], + [ + "Ġp", + "u" + ], + [ + "m", + "a" + ], + [ + "Ġs", + "l" + ], + [ + "Ġs", + "or" + ], + [ + "Ġm", + "o" + ], + [ + "v", + "en" + ], + [ + "is", + "h" + ], + [ + "Ġsh", + "ow" + ], + [ + "Ġcould", + "n" + ], + [ + "au", + "se" + ], + [ + "in", + "k" + ], + [ + "B", + "ut" + ], + [ + "Ġh", + "ouse" + ], + [ + "Ġint", + "o" + ], + [ + "um", + "p" + ], + [ + "Ġo", + "ver" + ], + [ + "Ġt", + "ried" + ], + [ + "a", + "nd" + ], + [ + "Ġe", + "at" + ], + [ + "Ġs", + "k" + ], + [ + "Ġs", + "un" + ], + [ + "Ġt", + "w" + ], + [ + "Ġcl", + "o" + ], + [ + "i", + "a" + ], + [ + "Ġr", + "un" + ], + [ + "Ġha", + "nd" + ], + [ + "ĠE", + "very" + ], + [ + "Ġ", + "en" + ], + [ + "Ġin", + "side" + ], + [ + "d", + "y" + ], + [ + "Ġ", + "if" + ], + [ + "Ġto", + "ld" + ], + [ + "Ġne", + "ver" + ], + [ + "i", + "on" + ], + [ + "Ġ", + "qu" + ], + [ + "Ġbec", + "ause" + ], + [ + "b", + "y" + ], + [ + "Ġpr", + "oud" + ], + [ + "Ġg", + "ave" + ], + [ + "Ġsor", + "ry" + ], + [ + "Ġth", + "is" + ], + [ + "Ġo", + "p" + ], + [ + "Ġplay", + "ed" + ], + [ + "at", + "e" + ], + [ + "Ġex", + "pl" + ], + [ + "Ġhe", + "ard" + ], + [ + "t", + "y" + ], + [ + "Ġo", + "r" + ], + [ + "Ġwa", + "ter" + ], + [ + "an", + "k" + ], + [ + "g", + "e" + ], + [ + "s", + "ed" + ], + [ + "Ġp", + "ick" + ], + [ + "ot", + "her" + ], + [ + "Ġro", + "om" + ], + [ + "et", + "ter" + ], + [ + "Ġj", + "ust" + ], + [ + "a", + "ce" + ], + [ + "Ġhug", + "ged" + ], + [ + "a", + "k" + ], + [ + "Ġg", + "re" + ], + [ + "he", + "re" + ], + [ + "Ġof", + "f" + ], + [ + "ĠS", + "ara" + ], + [ + "Ġp", + "ret" + ], + [ + "il", + "e" + ], + [ + "Ġe", + "ach" + ], + [ + "Ġc", + "om" + ], + [ + "Ġl", + "ong" + ], + [ + "Ġbo", + "x" + ], + [ + "or", + "t" + ], + [ + "Ġst", + "r" + ], + [ + "i", + "z" + ], + [ + "Ġu", + "nt" + ], + [ + "Ġwa", + "t" + ], + [ + "ot", + "h" + ], + [ + "Ġne", + "ed" + ], + [ + "Ġj", + "o" + ], + [ + "ĠW", + "e" + ], + [ + "T", + "om" + ], + [ + "Ġsm", + "all" + ], + [ + "in", + "e" + ], + [ + "Ġbe", + "ar" + ], + [ + "M", + "om" + ], + [ + "Ġunt", + "il" + ], + [ + "Ġn", + "ice" + ], + [ + "Ġt", + "ry" + ], + [ + "v", + "ing" + ], + [ + "u", + "c" + ], + [ + "s", + "el" + ], + [ + "oug", + "h" + ], + [ + "Ġlearn", + "ed" + ], + [ + "Ġk", + "ind" + ], + [ + "ĠA", + "nna" + ], + [ + "il", + "d" + ], + [ + "Ġf", + "o" + ], + [ + "Ġman", + "y" + ], + [ + "'", + "m" + ], + [ + "ĠJ", + "ack" + ], + [ + "Ġb", + "etter" + ], + [ + "Ġ", + "im" + ], + [ + "g", + "ry" + ], + [ + "im", + "al" + ], + [ + "Ġan", + "imal" + ], + [ + "ur", + "t" + ], + [ + "f", + "t" + ], + [ + "Ġe", + "nd" + ], + [ + "Ġs", + "n" + ], + [ + "a", + "ut" + ], + [ + "Ġt", + "e" + ], + [ + "Ġc", + "le" + ], + [ + "ĠJ", + "o" + ], + [ + "v", + "ent" + ], + [ + "ur", + "p" + ], + [ + "Ġg", + "r" + ], + [ + "Ġbe", + "aut" + ], + [ + "Ġj", + "ump" + ], + [ + "m", + "b" + ], + [ + "Ġa", + "d" + ], + [ + "re", + "am" + ], + [ + "p", + "t" + ], + [ + "ĠS", + "o" + ], + [ + "ĠH", + "er" + ], + [ + "Ġfl", + "ow" + ], + [ + "i", + "es" + ], + [ + "Ġc", + "he" + ], + [ + "Ġb", + "ra" + ], + [ + "Ġthan", + "ked" + ], + [ + "Ġe", + "ven" + ], + [ + "Ġb", + "est" + ], + [ + "Ġc", + "all" + ], + [ + "ad", + "y" + ], + [ + "H", + "e" + ], + [ + "Ġlot", + "s" + ], + [ + "Ġlaug", + "hed" + ], + [ + "sel", + "f" + ], + [ + "Ġr", + "a" + ], + [ + "Ġwa", + "y" + ], + [ + "ar", + "s" + ], + [ + "ur", + "n" + ], + [ + "Ġb", + "y" + ], + [ + "Ġfa", + "st" + ], + [ + "ll", + "y" + ], + [ + "Ġf", + "am" + ], + [ + "Ġ", + "C" + ], + [ + "v", + "es" + ], + [ + "ard", + "en" + ], + [ + "Ġg", + "arden" + ], + [ + "Ġbeaut", + "i" + ], + [ + "w", + "n" + ], + [ + "T", + "h" + ], + [ + "Ġbeauti", + "ful" + ], + [ + "Ġl", + "oud" + ], + [ + "le", + "w" + ], + [ + "Ġsk", + "y" + ], + [ + "Ġd", + "on" + ], + [ + "h", + "n" + ], + [ + "er", + "ed" + ], + [ + "in", + "y" + ], + [ + "Ġcare", + "ful" + ], + [ + "Ġlo", + "ve" + ], + [ + "Ġf", + "i" + ], + [ + "ĠThe", + "n" + ], + [ + "Å", + "ĵ" + ], + [ + "a", + "se" + ], + [ + "ec", + "t" + ], + [ + "Ġsa", + "fe" + ], + [ + "ĠA", + "nd" + ], + [ + "Ġu", + "nder" + ], + [ + "Ġc", + "ome" + ], + [ + "ĠF", + "rom" + ], + [ + "Y", + "es" + ], + [ + "ĠM", + "ia" + ], + [ + "I", + "t" + ], + [ + "m", + "e" + ], + [ + "Ġh", + "ard" + ], + [ + "Ġc", + "u" + ], + [ + "Ġw", + "o" + ], + [ + "Ġl", + "ist" + ], + [ + "Ġst", + "ay" + ], + [ + "an", + "e" + ], + [ + "s", + "h" + ], + [ + "op", + "le" + ], + [ + "Ġg", + "l" + ], + [ + "n", + "ing" + ], + [ + "Ġst", + "ill" + ], + [ + "oo", + "l" + ], + [ + "Ġh", + "urt" + ], + [ + "re", + "e" + ], + [ + "ĠH", + "is" + ], + [ + "Ġim", + "p" + ], + [ + "Ġfam", + "ily" + ], + [ + "Ġ", + "â" + ], + [ + "Ġb", + "oth" + ], + [ + "r", + "m" + ], + [ + "ig", + "h" + ], + [ + "Ġli", + "ved" + ], + [ + "he", + "s" + ], + [ + "W", + "hen" + ], + [ + "Ġpe", + "ople" + ], + [ + "Ġanimal", + "s" + ], + [ + "Ġco", + "l" + ], + [ + "Ġbra", + "ve" + ], + [ + "Ġwal", + "ked" + ], + [ + "o", + "b" + ], + [ + "T", + "im" + ], + [ + "c", + "t" + ], + [ + "Ġl", + "et" + ], + [ + "urp", + "r" + ], + [ + "ĠW", + "hen" + ], + [ + "Ġtw", + "o" + ], + [ + "Ġs", + "urpr" + ], + [ + "Ġsh", + "ould" + ], + [ + "is", + "hed" + ], + [ + "Ġb", + "ad" + ], + [ + "re", + "ss" + ], + [ + "Ġke", + "pt" + ], + [ + "Ġf", + "ore" + ], + [ + "Ġf", + "lew" + ], + [ + "Ġf", + "in" + ], + [ + "Ġst", + "or" + ], + [ + "Ġf", + "ly" + ], + [ + "a", + "st" + ], + [ + "is", + "ed" + ], + [ + "i", + "p" + ], + [ + "Ġit", + "s" + ], + [ + "l", + "ed" + ], + [ + "o", + "ck" + ], + [ + "uc", + "y" + ], + [ + "f", + "ore" + ], + [ + "Ġgo", + "ing" + ], + [ + "Ġcle", + "an" + ], + [ + "Ġd", + "an" + ], + [ + "Ġp", + "ic" + ], + [ + "Ġso", + "on" + ], + [ + "Ġcall", + "ed" + ], + [ + "Ġsh", + "are" + ], + [ + "k", + "ay" + ], + [ + "Ġan", + "gry" + ], + [ + "Ġro", + "ck" + ], + [ + "Ġc", + "on" + ], + [ + "Ġpret", + "ty" + ], + [ + "N", + "o" + ], + [ + "Ġ", + "ide" + ], + [ + "i", + "ed" + ], + [ + "il", + "ly" + ], + [ + "Ġg", + "round" + ], + [ + "x", + "t" + ], + [ + "Ġr", + "ed" + ], + [ + "Ġexpl", + "ore" + ], + [ + "Ġc", + "ry" + ], + [ + "Ġad", + "vent" + ], + [ + "Ġst", + "o" + ], + [ + "s", + "o" + ], + [ + "Ġre", + "al" + ], + [ + "L", + "et" + ], + [ + "Ġw", + "ind" + ], + [ + "Ġsh", + "iny" + ], + [ + "b", + "e" + ], + [ + "Ġb", + "ook" + ], + [ + "Ġal", + "so" + ], + [ + "Ġdo", + "ll" + ], + [ + "Ġide", + "a" + ], + [ + "Ġbe", + "fore" + ], + [ + "Ġop", + "ened" + ], + [ + "dd", + "ed" + ], + [ + "Ġwh", + "ile" + ], + [ + "um", + "my" + ], + [ + "Ġke", + "ep" + ], + [ + "Ġe", + "y" + ], + [ + "Ġn", + "ow" + ], + [ + "Ġdo", + "or" + ], + [ + "Ġfeel", + "ing" + ], + [ + "âĤ¬", + "â" + ], + [ + "o", + "on" + ], + [ + "o", + "y" + ], + [ + "Ġwal", + "king" + ], + [ + "Ġno", + "ise" + ], + [ + "Ġf", + "r" + ], + [ + "le", + "s" + ], + [ + "ag", + "e" + ], + [ + "i", + "ous" + ], + [ + "Ġcol", + "or" + ], + [ + "Ġt", + "urn" + ], + [ + "t", + "hing" + ], + [ + "f", + "f" + ], + [ + "u", + "ch" + ], + [ + "t", + "h" + ], + [ + "Ġb", + "ed" + ], + [ + "ar", + "y" + ], + [ + "Ġd", + "ra" + ], + [ + "Ġpick", + "ed" + ], + [ + "im", + "b" + ], + [ + "e", + "et" + ], + [ + "Ġcl", + "imb" + ], + [ + "Ġd", + "el" + ], + [ + "W", + "hat" + ], + [ + "Ġbe", + "ing" + ], + [ + "Ġf", + "ood" + ], + [ + "Ġu", + "n" + ], + [ + "Ġf", + "ar" + ], + [ + "t", + "ure" + ], + [ + "j", + "oy" + ], + [ + "Ġadvent", + "ure" + ], + [ + "a", + "c" + ], + [ + "Ġsmil", + "e" + ], + [ + "Ġd", + "if" + ], + [ + "Ħ", + "¢" + ], + [ + "âĤ¬â", + "Ħ¢" + ], + [ + "me", + "mb" + ], + [ + "Ġw", + "r" + ], + [ + "Ġth", + "r" + ], + [ + "ug", + "ht" + ], + [ + "Ġloo", + "king" + ], + [ + "Ġne", + "xt" + ], + [ + "ic", + "ed" + ], + [ + "ĠL", + "ucy" + ], + [ + "Ġno", + "dded" + ], + [ + "Ġqu", + "ick" + ], + [ + "Ġ", + "P" + ], + [ + "Ġd", + "is" + ], + [ + "Ġre", + "pl" + ], + [ + "ĠD", + "ad" + ], + [ + "Ġwa", + "it" + ], + [ + "iz", + "ed" + ], + [ + "Ġfore", + "st" + ], + [ + "Ġclo", + "s" + ], + [ + "ĠS", + "uddenly" + ], + [ + "Ġt", + "ra" + ], + [ + "T", + "hat" + ], + [ + "Ġey", + "es" + ], + [ + "g", + "er" + ], + [ + "bb", + "it" + ], + [ + "t", + "ed" + ], + [ + "Ġo", + "wn" + ], + [ + "Ġr", + "ain" + ], + [ + "Ġimp", + "ort" + ], + [ + "Ġgre", + "at" + ], + [ + "Ġre", + "memb" + ], + [ + "Ġpic", + "ture" + ], + [ + "Th", + "ank" + ], + [ + "S", + "uddenly" + ], + [ + "Ġsto", + "pped" + ], + [ + "Ġen", + "joy" + ], + [ + "Ġv", + "o" + ], + [ + "B", + "en" + ], + [ + "Ġg", + "ive" + ], + [ + "Ġimport", + "ant" + ], + [ + "Ġwor", + "k" + ], + [ + "Ġne", + "ar" + ], + [ + "g", + "an" + ], + [ + "p", + "ot" + ], + [ + "Ġe", + "ver" + ], + [ + "Ġa", + "pp" + ], + [ + "Ġa", + "fter" + ], + [ + "Ġquick", + "ly" + ], + [ + "Ġlist", + "en" + ], + [ + "Ġb", + "re" + ], + [ + "t", + "ing" + ], + [ + "bb", + "ed" + ], + [ + "Ġm", + "a" + ], + [ + "Ġf", + "ish" + ], + [ + "Ġrepl", + "ied" + ], + [ + "Ġra", + "bbit" + ], + [ + "Ġhand", + "s" + ], + [ + "Ġ", + "G" + ], + [ + "Ġnot", + "iced" + ], + [ + "Ġbr", + "o" + ], + [ + "Ġsl", + "ide" + ], + [ + "Ġth", + "ink" + ], + [ + "Ġwal", + "k" + ], + [ + "Ġtr", + "uck" + ], + [ + "Ġa", + "c" + ], + [ + "k", + "es" + ], + [ + "Ġstr", + "ong" + ], + [ + "S", + "he" + ], + [ + "Ġshow", + "ed" + ], + [ + "Ġd", + "e" + ], + [ + "Ġevery", + "one" + ], + [ + "Ġwo", + "nder" + ], + [ + "fe", + "re" + ], + [ + "by", + "e" + ], + [ + "Ġdif", + "fere" + ], + [ + "ir", + "st" + ], + [ + "S", + "o" + ], + [ + "Ġs", + "ure" + ], + [ + "Ġha", + "s" + ], + [ + "Ġr", + "ight" + ], + [ + "Ġbe", + "en" + ], + [ + "Ġbec", + "ame" + ], + [ + "Ġs", + "ound" + ], + [ + "ma", + "z" + ], + [ + "Ġto", + "w" + ], + [ + "Ġr", + "u" + ], + [ + "Ġt", + "al" + ], + [ + "Ġa", + "maz" + ], + [ + "Ġhe", + "ad" + ], + [ + "Ġsh", + "out" + ], + [ + "Ġbr", + "ight" + ], + [ + "Ġy", + "e" + ], + [ + "Ġwat", + "ched" + ], + [ + "Ġ", + "R" + ], + [ + "Ġm", + "or" + ], + [ + "Ġch", + "ild" + ], + [ + "a", + "ble" + ], + [ + "Ġme", + "an" + ], + [ + "Ġw", + "here" + ], + [ + "ll", + "ow" + ], + [ + "Ġh", + "igh" + ], + [ + "ĠS", + "ue" + ], + [ + "Ġfa", + "ce" + ], + [ + "Ġc", + "ook" + ], + [ + "d", + "ay" + ], + [ + "ay", + "be" + ], + [ + "Ġwat", + "ch" + ], + [ + "Ġbl", + "ue" + ], + [ + "a", + "ught" + ], + [ + "Ġdiffere", + "nt" + ], + [ + "Ġst", + "ore" + ], + [ + "Ġ", + "N" + ], + [ + "Ġgood", + "bye" + ], + [ + "Ġd", + "ress" + ], + [ + "u", + "ll" + ], + [ + "ĠB", + "ob" + ], + [ + "n", + "g" + ], + [ + "an", + "ge" + ], + [ + "Ġs", + "qu" + ], + [ + "Ġo", + "kay" + ], + [ + "is", + "y" + ], + [ + "le", + "ase" + ], + [ + "ĠMom", + "my" + ], + [ + "Ġvo", + "ice" + ], + [ + "J", + "o" + ], + [ + "at", + "h" + ], + [ + "Ġn", + "ight" + ], + [ + "ĠS", + "pot" + ], + [ + "Ġu", + "s" + ], + [ + "Ġbo", + "at" + ], + [ + "Ġflow", + "ers" + ], + [ + "Ġpl", + "ace" + ], + [ + "Ġfo", + "llow" + ], + [ + "Ġa", + "r" + ], + [ + "Ġu", + "se" + ], + [ + "Ġclos", + "er" + ], + [ + "un", + "ny" + ], + [ + "le", + "ep" + ], + [ + "ir", + "ed" + ], + [ + "Ġfa", + "v" + ], + [ + "Ġy", + "ell" + ], + [ + "Ġgra", + "bbed" + ], + [ + "Ġcu", + "ri" + ], + [ + "Ġwa", + "rm" + ], + [ + "Ġc", + "r" + ], + [ + "Ġfor", + "g" + ], + [ + "ĠSara", + "h" + ], + [ + "ĠJo", + "hn" + ], + [ + "Ġm", + "ag" + ], + [ + "Ġst", + "ick" + ], + [ + "W", + "e" + ], + [ + "Ġjump", + "ed" + ], + [ + "Ġc", + "ake" + ], + [ + "m", + "ore" + ], + [ + "Ġt", + "ell" + ], + [ + "Ġany", + "more" + ], + [ + "A", + "fter" + ], + [ + "Ġbut", + "ter" + ], + [ + "nd", + "ma" + ], + [ + "Ġth", + "ree" + ], + [ + "Ġas", + "k" + ], + [ + "c", + "o" + ], + [ + "Ġ", + "our" + ], + [ + "l", + "ie" + ], + [ + "Ġcuri", + "ous" + ], + [ + "ou", + "nt" + ], + [ + "or", + "n" + ], + [ + "Ġcon", + "t" + ], + [ + "Ġfe", + "ll" + ], + [ + "a", + "ched" + ], + [ + "ĠT", + "h" + ], + [ + "Ġbird", + "s" + ], + [ + "as", + "s" + ], + [ + "H", + "er" + ], + [ + "is", + "s" + ], + [ + "Ġhelp", + "ed" + ], + [ + "ĠJ", + "ane" + ], + [ + "Ġpu", + "ll" + ], + [ + "Ġf", + "irst" + ], + [ + "it", + "c" + ], + [ + "Ġbl", + "ock" + ], + [ + "Ġh", + "op" + ], + [ + "Ġb", + "it" + ], + [ + "Ġd", + "r" + ], + [ + "Ġreal", + "ized" + ], + [ + "Ġk", + "id" + ], + [ + "L", + "ook" + ], + [ + "il", + "a" + ], + [ + "Ġm", + "on" + ], + [ + "Ġbr", + "other" + ], + [ + "A", + "nna" + ], + [ + "S", + "ara" + ], + [ + "Ġ", + "z" + ], + [ + "ĠEvery", + "one" + ], + [ + "Ġat", + "e" + ], + [ + "Ġdo", + "es" + ], + [ + "im", + "es" + ], + [ + "Ġhapp", + "ened" + ], + [ + "Ġst", + "op" + ], + [ + "z", + "y" + ], + [ + "Ġy", + "ummy" + ], + [ + "Ġfav", + "or" + ], + [ + "pp", + "y" + ], + [ + "Ġk", + "itc" + ], + [ + "Ġkitc", + "hen" + ], + [ + "Ġsw", + "eet" + ], + [ + "u", + "s" + ], + [ + "Ġp", + "er" + ], + [ + "Ġre", + "ally" + ], + [ + "a", + "isy" + ], + [ + "Ġgra", + "ss" + ], + [ + "Ġfavor", + "ite" + ], + [ + "Ġbe", + "gan" + ], + [ + "Ġre", + "st" + ], + [ + "Ġre", + "ady" + ], + [ + "Ġle", + "a" + ], + [ + "Ġre", + "ached" + ], + [ + "Ġunder", + "st" + ], + [ + "a", + "ir" + ], + [ + "Ġst", + "e" + ], + [ + "Ġb", + "unny" + ], + [ + "ĠA", + "s" + ], + [ + "Ġstor", + "y" + ], + [ + "'", + "re" + ], + [ + "Ġp", + "ain" + ], + [ + "Ġs", + "ing" + ], + [ + "st", + "er" + ], + [ + "Ġhe", + "re" + ], + [ + "Ġsa", + "nd" + ], + [ + "Ġon", + "ly" + ], + [ + "Ġfl", + "o" + ], + [ + "Ġa", + "m" + ], + [ + "Ġgl", + "ad" + ], + [ + "Ġt", + "ri" + ], + [ + "Ġbe", + "h" + ], + [ + "Ġwor", + "ld" + ], + [ + "Ġop", + "en" + ], + [ + "Ġc", + "re" + ], + [ + "fu", + "lly" + ], + [ + "Ġpr", + "in" + ], + [ + "w", + "here" + ], + [ + "are", + "nt" + ], + [ + "Ġflow", + "er" + ], + [ + "Ġthr", + "ough" + ], + [ + "Ġb", + "a" + ], + [ + "Ġfi", + "re" + ], + [ + "Ġd", + "one" + ], + [ + "Ġha", + "ving" + ], + [ + "Ġth", + "ing" + ], + [ + "Ġdel", + "ic" + ], + [ + "Ġhim", + "self" + ], + [ + "Ġt", + "ired" + ], + [ + "Ġp", + "arent" + ], + [ + "Ġso", + "ft" + ], + [ + "Ġf", + "ro" + ], + [ + "Tim", + "my" + ], + [ + "Ġta", + "st" + ], + [ + "Ġbutter", + "f" + ], + [ + "ĠL", + "et" + ], + [ + "Ġc", + "ut" + ], + [ + "Ġp", + "art" + ], + [ + "Ġwh", + "y" + ], + [ + "k", + "en" + ], + [ + "Ġm", + "ess" + ], + [ + "C", + "an" + ], + [ + "Ġwor", + "ry" + ], + [ + "Mom", + "my" + ], + [ + "i", + "ver" + ], + [ + "Ġd", + "in" + ], + [ + "u", + "ff" + ], + [ + "at", + "er" + ], + [ + "Ġmag", + "ic" + ], + [ + "Ġwa", + "ved" + ], + [ + "Ġshout", + "ed" + ], + [ + "Ġpo", + "nd" + ], + [ + "Ġkid", + "s" + ], + [ + "Ġh", + "at" + ], + [ + "Ġd", + "uck" + ], + [ + "Ġse", + "es" + ], + [ + "o", + "lly" + ], + [ + "ill", + "ed" + ], + [ + "Ġg", + "ame" + ], + [ + "i", + "ent" + ], + [ + "Ġma", + "king" + ], + [ + "at", + "her" + ], + [ + "Jo", + "hn" + ], + [ + "A", + "s" + ], + [ + "ak", + "es" + ], + [ + "Ġcat", + "ch" + ], + [ + "Ġse", + "en" + ], + [ + "Ġc", + "ool" + ], + [ + "at", + "ion" + ], + [ + "Ġcom", + "ing" + ], + [ + "Ġl", + "ess" + ], + [ + "Ġd", + "ark" + ], + [ + "Ġu", + "sed" + ], + [ + "ed", + "dy" + ], + [ + "Ġfi", + "x" + ], + [ + "ĠJo", + "e" + ], + [ + "Ġthan", + "k" + ], + [ + "m", + "er" + ], + [ + "Ġto", + "p" + ], + [ + "Ġl", + "ady" + ], + [ + "Ġha", + "ir" + ], + [ + "ar", + "ing" + ], + [ + "Ġp", + "rom" + ], + [ + "Ġho", + "pped" + ], + [ + "ĠC", + "an" + ], + [ + "\"", + "." + ], + [ + "ig", + "n" + ], + [ + "Ġsurpr", + "ise" + ], + [ + "Ġdra", + "w" + ], + [ + "Ġm", + "um" + ], + [ + "Ġm", + "ouse" + ], + [ + "Ġfun", + "ny" + ], + [ + "re", + "l" + ], + [ + "Ġc", + "ra" + ], + [ + "Ġ", + "-" + ], + [ + "ap", + "er" + ], + [ + "Ġf", + "ull" + ], + [ + "Ġto", + "uch" + ], + [ + "Ġl", + "ight" + ], + [ + "Ġsp", + "ot" + ], + [ + "re", + "n" + ], + [ + "Ġd", + "ro" + ], + [ + "ĠBen", + "ny" + ], + [ + "Ġparent", + "s" + ], + [ + "Ġwor", + "ked" + ], + [ + "in", + "s" + ], + [ + "one", + "y" + ], + [ + "ĠD", + "o" + ], + [ + "Ġsurpr", + "ised" + ], + [ + "Ġcare", + "fully" + ], + [ + "Ġtre", + "es" + ], + [ + "Ġfro", + "g" + ], + [ + "Ġsw", + "ing" + ], + [ + "Ġdo", + "ing" + ], + [ + "D", + "on" + ], + [ + "Ġs", + "at" + ], + [ + "in", + "ally" + ], + [ + "ĠT", + "hat" + ], + [ + "Ġ", + "ice" + ], + [ + "ĠI", + "n" + ], + [ + "Ġhe", + "ld" + ], + [ + "W", + "ow" + ], + [ + "Ġrun", + "ning" + ], + [ + "Ġpret", + "end" + ], + [ + "Ġsw", + "im" + ], + [ + "Ġs", + "et" + ], + [ + "Ġre", + "ad" + ], + [ + "Ġdelic", + "ious" + ], + [ + "Ġneed", + "ed" + ], + [ + "Ġt", + "ight" + ], + [ + "Ġsl", + "ow" + ], + [ + "Ġrememb", + "ered" + ], + [ + "Ġlo", + "st" + ], + [ + "Ġco", + "ld" + ], + [ + "Ġsm", + "ell" + ], + [ + "ard", + "s" + ], + [ + "Ġw", + "ood" + ], + [ + "Ġhapp", + "ily" + ], + [ + "h", + "y" + ], + [ + "el", + "y" + ], + [ + "Ġlook", + "s" + ], + [ + "Ġbeh", + "ind" + ], + [ + "Ġher", + "self" + ], + [ + "Ġc", + "ried" + ], + [ + "Ġenjoy", + "ed" + ], + [ + "Ġnam", + "e" + ], + [ + "as", + "k" + ], + [ + "Ġye", + "ars" + ], + [ + "Ġbu", + "y" + ], + [ + "Ġho", + "le" + ], + [ + "Ġblock", + "s" + ], + [ + "Ġd", + "ri" + ], + [ + "Ġs", + "leep" + ], + [ + "Ġg", + "i" + ], + [ + "Ġyell", + "ow" + ], + [ + "c", + "ess" + ], + [ + "Ġbutterf", + "ly" + ], + [ + "i", + "ke" + ], + [ + "u", + "ed" + ], + [ + "Ġw", + "ished" + ], + [ + "Ġper", + "f" + ], + [ + "Ġan", + "other" + ], + [ + "ĠD", + "aisy" + ], + [ + "Ġgre", + "en" + ], + [ + "Ġmo", + "ve" + ], + [ + "Ġt", + "all" + ], + [ + "Ġa", + "ir" + ], + [ + "Ġb", + "ag" + ], + [ + "Ġflo", + "or" + ], + [ + "Ġcar", + "s" + ], + [ + "Ġless", + "on" + ], + [ + "u", + "l" + ], + [ + "Ġb", + "ow" + ], + [ + "Ġar", + "ri" + ], + [ + "Ġwind", + "ow" + ], + [ + "Ġsh", + "o" + ], + [ + "Ġchild", + "ren" + ], + [ + "n", + "er" + ], + [ + "Ġfin", + "ished" + ], + [ + "Ġh", + "ill" + ], + [ + "ĠA", + "fter" + ], + [ + "and", + "y" + ], + [ + "s", + "p" + ], + [ + "en", + "s" + ], + [ + "Ġp", + "aper" + ], + [ + "Ġli", + "kes" + ], + [ + "F", + "rom" + ], + [ + "s", + "y" + ], + [ + "Ġho", + "ld" + ], + [ + "S", + "am" + ], + [ + "Ġwr", + "ong" + ], + [ + "re", + "ed" + ], + [ + "Ġcont", + "in" + ], + [ + "Ġunderst", + "and" + ], + [ + "at", + "ter" + ], + [ + "u", + "it" + ], + [ + "re", + "w" + ], + [ + "Ġg", + "ent" + ], + [ + "Ġany", + "thing" + ], + [ + "Ġclo", + "se" + ], + [ + "Ġwa", + "ll" + ], + [ + "Ġe", + "l" + ], + [ + "as", + "ure" + ], + [ + "Ġle", + "ft" + ], + [ + "Ġa", + "ble" + ], + [ + "Ġarri", + "ved" + ], + [ + "Ġh", + "un" + ], + [ + "ro", + "ss" + ], + [ + "a", + "v" + ], + [ + "Ġhe", + "ar" + ], + [ + "Ġfriend", + "ly" + ], + [ + "Ġforg", + "ot" + ], + [ + "Ġg", + "one" + ], + [ + "am", + "a" + ], + [ + "Ġcre", + "at" + ], + [ + "Ġw", + "et" + ], + [ + "Ġli", + "on" + ], + [ + "H", + "ell" + ], + [ + "Hell", + "o" + ], + [ + "Ġd", + "ream" + ], + [ + "Ġh", + "ot" + ], + [ + "b", + "o" + ], + [ + "b", + "er" + ], + [ + "ĠS", + "ally" + ], + [ + "Ġf", + "illed" + ], + [ + "Ġd", + "ir" + ], + [ + "ie", + "ld" + ], + [ + "Ġcook", + "ies" + ], + [ + "Ġlaug", + "h" + ], + [ + "Ġbro", + "ken" + ], + [ + "Ġdin", + "ner" + ], + [ + "l", + "f" + ], + [ + "Ġel", + "se" + ], + [ + "Ġh", + "id" + ], + [ + "c", + "ed" + ], + [ + "Ġp", + "ink" + ], + [ + "Ġfollow", + "ed" + ], + [ + "Ġta", + "ble" + ], + [ + "Ġm", + "ar" + ], + [ + "Ġcolor", + "s" + ], + [ + "ou", + "p" + ], + [ + "Ġmom", + "ent" + ], + [ + "Ġcontin", + "ued" + ], + [ + "Ġwonder", + "ful" + ], + [ + "ĠE", + "m" + ], + [ + "e", + "y" + ], + [ + "in", + "ed" + ], + [ + "ir", + "rel" + ], + [ + "r", + "ot" + ], + [ + "Ġfin", + "ally" + ], + [ + "J", + "ack" + ], + [ + "Ġa", + "rm" + ], + [ + "Ġm", + "ight" + ], + [ + "ĠN", + "ow" + ], + [ + "Ġc", + "ast" + ], + [ + "Ġy", + "es" + ], + [ + "Ġbu", + "ild" + ], + [ + "Ġen", + "ough" + ], + [ + "ĠB", + "illy" + ], + [ + "Ġsome", + "one" + ], + [ + "Ġperf", + "ect" + ], + [ + "Ġfa", + "ir" + ], + [ + "Ġstay", + "ed" + ], + [ + "a", + "pp" + ], + [ + "Ġmor", + "ning" + ], + [ + "ĠThe", + "re" + ], + [ + "Ġpu", + "ppy" + ], + [ + "o", + "l" + ], + [ + "Ġdad", + "dy" + ], + [ + "Ġtry", + "ing" + ], + [ + "'", + "ll" + ], + [ + "Ġj", + "u" + ], + [ + "Ġother", + "s" + ], + [ + "Ġbook", + "s" + ], + [ + "Ġag", + "reed" + ], + [ + "Ġmu", + "s" + ], + [ + "Ġsn", + "ow" + ], + [ + "Ġsqu", + "irrel" + ], + [ + "Ġc", + "ream" + ], + [ + "ro", + "om" + ], + [ + "Ġget", + "ting" + ], + [ + "Ġgra", + "ndma" + ], + [ + "ĠL", + "ila" + ], + [ + "Ġlea", + "ves" + ], + [ + "Ġba", + "by" + ], + [ + "Ġc", + "ount" + ], + [ + "ic", + "y" + ], + [ + "Ġfe", + "w" + ], + [ + "A", + "t" + ], + [ + "Ġf", + "all" + ], + [ + "The", + "n" + ], + [ + "g", + "on" + ], + [ + "Ġbre", + "ak" + ], + [ + "Ġclimb", + "ed" + ], + [ + "ri", + "es" + ], + [ + "Ġbe", + "ach" + ], + [ + "as", + "h" + ], + [ + "Ġb", + "ug" + ], + [ + "Ġp", + "lease" + ], + [ + "Ġm", + "other" + ], + [ + "Ġbr", + "ought" + ], + [ + "Ġtra", + "in" + ], + [ + "an", + "ce" + ], + [ + "ĠTh", + "is" + ], + [ + "Ġde", + "ep" + ], + [ + "Ġv", + "is" + ], + [ + "at", + "ed" + ], + [ + "s", + "hed" + ], + [ + "Ġm", + "et" + ], + [ + "Ġwas", + "n" + ], + [ + "Ġsomet", + "imes" + ], + [ + "Ġevery", + "thing" + ], + [ + "ĠTom", + "my" + ], + [ + "i", + "ch" + ], + [ + "Ġmus", + "ic" + ], + [ + "ke", + "y" + ], + [ + "l", + "ing" + ], + [ + "Th", + "is" + ], + [ + "o", + "s" + ], + [ + "Ġturn", + "ed" + ], + [ + "l", + "c" + ], + [ + "Ġr", + "ide" + ], + [ + "o", + "om" + ], + [ + "Ġs", + "ong" + ], + [ + "or", + "ing" + ], + [ + "Ġsp", + "ark" + ], + [ + "Ġf", + "ight" + ], + [ + "Ġhun", + "gry" + ], + [ + "on", + "s" + ], + [ + "Ġpicture", + "s" + ], + [ + "d", + "s" + ], + [ + "Ġm", + "akes" + ], + [ + "Ġse", + "ar" + ], + [ + "uck", + "y" + ], + [ + "Ġlist", + "ened" + ], + [ + "Ġjo", + "y" + ], + [ + "Ġpo", + "l" + ], + [ + "ĠW", + "hat" + ], + [ + "H", + "is" + ], + [ + "e", + "e" + ], + [ + "Ġcl", + "ot" + ], + [ + "Ġsa", + "il" + ], + [ + "a", + "pped" + ], + [ + "b", + "ble" + ], + [ + "Ġcom", + "p" + ], + [ + "Ġpl", + "an" + ], + [ + "Ġwant", + "s" + ], + [ + "Ġclot", + "hes" + ], + [ + "Ġpo", + "in" + ], + [ + "Ġ", + "K" + ], + [ + "z", + "z" + ], + [ + "Ġball", + "oon" + ], + [ + "Ġstr", + "ange" + ], + [ + "m", + "an" + ], + [ + "en", + "ny" + ], + [ + "Ġp", + "i" + ], + [ + "Ġgr", + "ow" + ], + [ + "Ġfly", + "ing" + ], + [ + "r", + "ow" + ], + [ + "Ġsc", + "ary" + ], + [ + "Ġtre", + "at" + ], + [ + "Ġwa", + "ited" + ], + [ + "Ġamaz", + "ed" + ], + [ + "Ġt", + "eddy" + ], + [ + "Ġcast", + "le" + ], + [ + "is", + "ter" + ], + [ + "Ġe", + "le" + ], + [ + "Ġamaz", + "ing" + ], + [ + "m", + "ed" + ], + [ + "Ġsp", + "l" + ], + [ + "O", + "h" + ], + [ + "Ġle", + "ave" + ], + [ + "Ġown", + "er" + ], + [ + "Ġprom", + "ised" + ], + [ + "Ġdan", + "ger" + ], + [ + "e", + "ver" + ], + [ + "O", + "K" + ], + [ + "Ġs", + "uch" + ], + [ + "Ġwh", + "ite" + ], + [ + "Ġp", + "at" + ], + [ + "Ġtal", + "k" + ], + [ + "uff", + "y" + ], + [ + "Ġf", + "ur" + ], + [ + "Ġp", + "ie" + ], + [ + "op", + "e" + ], + [ + "r", + "ed" + ], + [ + "Ġpull", + "ed" + ], + [ + "Ġp", + "re" + ], + [ + "Ġwood", + "s" + ], + [ + "ĠT", + "o" + ], + [ + "Ġsh", + "ared" + ], + [ + "Ġal", + "one" + ], + [ + "Ġtow", + "ards" + ], + [ + "c", + "le" + ], + [ + "ĠM", + "aybe" + ], + [ + "Ġs", + "it" + ], + [ + "Ġn", + "ap" + ], + [ + "Ġcolor", + "ful" + ], + [ + "am", + "p" + ], + [ + "h", + "one" + ], + [ + "Ġvis", + "it" + ], + [ + "ig", + "g" + ], + [ + "ug", + "g" + ], + [ + "th", + "y" + ], + [ + "en", + "ce" + ], + [ + "Ġta", + "il" + ], + [ + "ct", + "or" + ], + [ + "Ġmagic", + "al" + ], + [ + "Ġr", + "iver" + ], + [ + "Ġh", + "ig" + ], + [ + "Ġbe", + "lie" + ], + [ + "Ġgr", + "ate" + ], + [ + "u", + "ally" + ], + [ + "an", + "g" + ], + [ + "ot", + "t" + ], + [ + "Ġh", + "ide" + ], + [ + "Ġs", + "illy" + ], + [ + "Ġsmil", + "es" + ], + [ + "ow", + "er" + ], + [ + "Ġs", + "ide" + ], + [ + "M", + "ax" + ], + [ + "Ġro", + "ll" + ], + [ + "Ġg", + "u" + ], + [ + "ĠA", + "my" + ], + [ + "Ġpr", + "o" + ], + [ + "et", + "er" + ], + [ + "Ġfast", + "er" + ], + [ + "Ġgrate", + "ful" + ], + [ + "Ġtre", + "asure" + ], + [ + "as", + "ed" + ], + [ + "Ġsn", + "ack" + ], + [ + "u", + "p" + ], + [ + "Ġla", + "nd" + ], + [ + "av", + "y" + ], + [ + "Ġmor", + "al" + ], + [ + "ain", + "ed" + ], + [ + "Ġs", + "ister" + ], + [ + "ĠG", + "ra" + ], + [ + "Ġsh", + "ap" + ], + [ + "Ġpain", + "t" + ], + [ + "Ġslow", + "ly" + ], + [ + "Ġf", + "ield" + ], + [ + "Ġp", + "et" + ], + [ + "Ġs", + "ec" + ], + [ + "or", + "m" + ], + [ + "ĠJ", + "im" + ], + [ + "Ġbr", + "an" + ], + [ + "Ġm", + "in" + ], + [ + "Ġmu", + "st" + ], + [ + "Ġhelp", + "ing" + ], + [ + "Ġwor", + "ried" + ], + [ + "Ġjo", + "b" + ], + [ + "Ġfo", + "x" + ], + [ + "s", + "et" + ], + [ + "Ġm", + "oney" + ], + [ + "Ġp", + "ot" + ], + [ + "Ġqu", + "i" + ], + [ + "Ġbe", + "l" + ], + [ + "Ġac", + "c" + ], + [ + "Ġli", + "ving" + ], + [ + "Ġmon", + "ster" + ], + [ + "le", + "ss" + ], + [ + "Ġpart", + "y" + ], + [ + "e", + "ar" + ], + [ + "Ġmo", + "st" + ], + [ + "Ġwe", + "ar" + ], + [ + "Ġrememb", + "er" + ], + [ + "i", + "re" + ], + [ + "Ġs", + "ign" + ], + [ + "w", + "l" + ], + [ + "Ġcl", + "oud" + ], + [ + "Ġc", + "andy" + ], + [ + "Ġhig", + "her" + ], + [ + "i", + "pped" + ], + [ + "id", + "ent" + ], + [ + "ĠA", + "ll" + ], + [ + "d", + "er" + ], + [ + "z", + "e" + ], + [ + "Ġqui", + "et" + ], + [ + "Ġto", + "wn" + ], + [ + "un", + "ch" + ], + [ + "Ġprin", + "cess" + ], + [ + "Ġrun", + "s" + ], + [ + "ask", + "et" + ], + [ + "e", + "m" + ], + [ + "Ġevery", + "where" + ], + [ + "Ġjo", + "in" + ], + [ + "Ġste", + "pped" + ], + [ + "Ġb", + "asket" + ], + [ + "Ġl", + "ake" + ], + [ + "Ġcu", + "p" + ], + [ + "s", + "w" + ], + [ + "out", + "h" + ], + [ + "Ġcar", + "rot" + ], + [ + "Ġco", + "ll" + ], + [ + "Ġpl", + "ant" + ], + [ + "Ġsc", + "ream" + ], + [ + "ĠDad", + "dy" + ], + [ + "Ġb", + "uck" + ], + [ + "Ġhe", + "avy" + ], + [ + "Ġbig", + "ger" + ], + [ + "Ġch", + "o" + ], + [ + "Ġbl", + "ank" + ], + [ + "Ġclo", + "sed" + ], + [ + "Ġw", + "on" + ], + [ + "is", + "es" + ], + [ + "Ġan", + "sw" + ], + [ + "b", + "r" + ], + [ + "Ġblank", + "et" + ], + [ + "Ġm", + "outh" + ], + [ + "am", + "es" + ], + [ + "ec", + "es" + ], + [ + "Ġeat", + "ing" + ], + [ + "Ġpi", + "eces" + ], + [ + "ck", + "et" + ], + [ + "Ġst", + "re" + ], + [ + "Ġwe", + "lc" + ], + [ + "Ġwould", + "n" + ], + [ + "Ġre", + "ach" + ], + [ + "Ġbro", + "ke" + ], + [ + "e", + "ared" + ], + [ + "Ġdanger", + "ous" + ], + [ + "g", + "s" + ], + [ + "p", + "h" + ], + [ + "ĠM", + "olly" + ], + [ + "Ġc", + "e" + ], + [ + "ĠM", + "r" + ], + [ + "or", + "d" + ], + [ + "f", + "ort" + ], + [ + "Ġdoll", + "s" + ], + [ + "m", + "ing" + ], + [ + "Ġdan", + "ce" + ], + [ + "Ġdra", + "gon" + ], + [ + "Ġthink", + "s" + ], + [ + "Ġask", + "s" + ], + [ + "Ġac", + "ross" + ], + [ + "O", + "kay" + ], + [ + "Ġch", + "air" + ], + [ + "Ġte", + "ac" + ], + [ + "Ġb", + "ott" + ], + [ + "H", + "i" + ], + [ + "Ġp", + "a" + ], + [ + "Ġne", + "igh" + ], + [ + "Ġb", + "ar" + ], + [ + "Ġb", + "ike" + ], + [ + "Ġp", + "ack" + ], + [ + "it", + "ing" + ], + [ + "Ġdis", + "app" + ], + [ + "Ġneigh", + "b" + ], + [ + "Ġst", + "uck" + ], + [ + "d", + "e" + ], + [ + "Ġg", + "if" + ], + [ + "Ġfr", + "uit" + ], + [ + "Ġbelie", + "ve" + ], + [ + "c", + "y" + ], + [ + "Ġ", + "ve" + ], + [ + "Ġcry", + "ing" + ], + [ + "i", + "er" + ], + [ + "ph", + "ant" + ], + [ + "Ġs", + "uddenly" + ], + [ + "Ġs", + "oup" + ], + [ + "Ġr", + "ace" + ], + [ + "Ġth", + "rew" + ], + [ + "S", + "ure" + ], + [ + "Ġfar", + "mer" + ], + [ + "Ġacc", + "ident" + ], + [ + "Ġdir", + "ty" + ], + [ + "Ġele", + "phant" + ], + [ + "Ġmon", + "key" + ], + [ + "Ġhe", + "al" + ], + [ + "w", + "eet" + ], + [ + "Ġre", + "m" + ], + [ + "Ġp", + "ers" + ], + [ + "Ġsa", + "me" + ], + [ + "or", + "ed" + ], + [ + "vent", + "ually" + ], + [ + "e", + "ad" + ], + [ + "pp", + "ing" + ], + [ + "Ġgent", + "le" + ], + [ + "Ġf", + "ree" + ], + [ + "Ġbr", + "own" + ], + [ + "M", + "aybe" + ], + [ + "Ġin", + "st" + ], + [ + "Ġbe", + "e" + ], + [ + "i", + "x" + ], + [ + "Ġb", + "en" + ], + [ + "Ġw", + "in" + ], + [ + "Ġbl", + "ack" + ], + [ + "Ġal", + "ong" + ], + [ + "Ġlong", + "er" + ], + [ + "i", + "ble" + ], + [ + "Ġsp", + "r" + ], + [ + "Ġcl", + "apped" + ], + [ + "F", + "inally" + ], + [ + "W", + "hy" + ], + [ + "ck", + "ed" + ], + [ + "Ġwor", + "ds" + ], + [ + "ĠF", + "l" + ], + [ + "Ġteac", + "her" + ], + [ + "Ġup", + "set" + ], + [ + "ch", + "ool" + ], + [ + "Ġpie", + "ce" + ], + [ + "Ġwe", + "ll" + ], + [ + "Ġche", + "ered" + ], + [ + "Ġh", + "it" + ], + [ + "Ġsa", + "ng" + ], + [ + "m", + "ent" + ], + [ + "Ġbow", + "l" + ], + [ + "Ġb", + "ite" + ], + [ + "Ġdo", + "ctor" + ], + [ + "ĠJ", + "ill" + ], + [ + "Ġbuck", + "et" + ], + [ + "M", + "ia" + ], + [ + "Ġb", + "ath" + ], + [ + "Ġc", + "aught" + ], + [ + "Ġhe", + "art" + ], + [ + "Ġfor", + "get" + ], + [ + "Ġm", + "ark" + ], + [ + "Ġbut", + "t" + ], + [ + "Ġd", + "ry" + ], + [ + "Ġyou", + "ng" + ], + [ + "Ġse", + "a" + ], + [ + "Ġc", + "our" + ], + [ + "Ġdro", + "pped" + ], + [ + "es", + "e" + ], + [ + "Ġy", + "ard" + ], + [ + "Ġpl", + "ac" + ], + [ + "our", + "n" + ], + [ + "Ġs", + "chool" + ], + [ + "Ġsw", + "ings" + ], + [ + "bb", + "y" + ], + [ + "Ġw", + "ings" + ], + [ + "b", + "s" + ], + [ + "Ġexpl", + "ained" + ], + [ + "Ġj", + "ourn" + ], + [ + "Ġbr", + "ing" + ], + [ + "Ġdr", + "ink" + ], + [ + "Ġstre", + "et" + ], + [ + "Ġju", + "ice" + ], + [ + "un", + "g" + ], + [ + "Ġnear", + "by" + ], + [ + "ĠEm", + "ma" + ], + [ + "Ġsm", + "o" + ], + [ + "Ġsm", + "art" + ], + [ + "Ġpu", + "shed" + ], + [ + "Ġstor", + "ies" + ], + [ + "Ġdro", + "ve" + ], + [ + "Ġt", + "iny" + ], + [ + "Ġp", + "en" + ], + [ + "Ġcour", + "se" + ], + [ + "Ġ", + "es" + ], + [ + "Ġm", + "ine" + ], + [ + "Ġto", + "day" + ], + [ + "Ġpo", + "cket" + ], + [ + "Ġye", + "ar" + ], + [ + "aught", + "er" + ], + [ + "Ġj", + "e" + ], + [ + "Ġsec", + "ret" + ], + [ + "Ġex", + "p" + ], + [ + "Ġwith", + "out" + ], + [ + "Ġsing", + "ing" + ], + [ + "Ġwelc", + "ome" + ], + [ + "Ġhapp", + "en" + ], + [ + "t", + "o" + ], + [ + "Ġf", + "it" + ], + [ + "Ġst", + "u" + ], + [ + "ve", + "l" + ], + [ + "ra", + "ct" + ], + [ + "Ġbu", + "s" + ], + [ + "Ġche", + "ese" + ], + [ + "Ġcr", + "ay" + ], + [ + "Ġfair", + "y" + ], + [ + "Ġj", + "ar" + ], + [ + "Ġturn", + "s" + ], + [ + "Ġbu", + "sh" + ], + [ + "b", + "ow" + ], + [ + "ll", + "o" + ], + [ + "Ġexpl", + "oring" + ], + [ + "Ġl", + "on" + ], + [ + "Ġst", + "and" + ], + [ + "itt", + "ing" + ], + [ + "Ġsee", + "med" + ], + [ + "Ġsp", + "oon" + ], + [ + "ĠF", + "inally" + ], + [ + "Ġg", + "ames" + ], + [ + "Ġwo", + "ke" + ], + [ + "is", + "sed" + ], + [ + "Ġre", + "sp" + ], + [ + "Ġtr", + "ou" + ], + [ + "Ġtw", + "ins" + ], + [ + "Ġhop", + "ed" + ], + [ + "Ġat", + "t" + ], + [ + "Ġlon", + "ely" + ], + [ + "Ġa", + "nt" + ], + [ + "ĠM", + "ama" + ], + [ + "or", + "row" + ], + [ + "Ġno", + "ises" + ], + [ + "Ġhug", + "s" + ], + [ + "ount", + "ain" + ], + [ + "y", + "ard" + ], + [ + "Ġd", + "aughter" + ], + [ + "Ġin", + "v" + ], + [ + "Ġke", + "y" + ], + [ + "a", + "il" + ], + [ + "Ġrock", + "s" + ], + [ + "Ġdis", + "co" + ], + [ + "Ġs", + "u" + ], + [ + "Ġl", + "unch" + ], + [ + "re", + "ad" + ], + [ + "ou", + "n" + ], + [ + "ver", + "ed" + ], + [ + "Ġback", + "yard" + ], + [ + "Ġs", + "uc" + ], + [ + "Ġc", + "ow" + ], + [ + "Ġapp", + "le" + ], + [ + "e", + "k" + ], + [ + "Ġsw", + "am" + ], + [ + "Ġsho", + "es" + ], + [ + "Ġst", + "ars" + ], + [ + "Ġsh", + "ook" + ], + [ + "itt", + "ens" + ], + [ + "Ġm", + "iss" + ], + [ + "c", + "hes" + ], + [ + "Ġw", + "ish" + ], + [ + "Ġre", + "lie" + ], + [ + "fu", + "sed" + ], + [ + "Ġshap", + "es" + ], + [ + "urp", + "le" + ], + [ + "Ġtow", + "er" + ], + [ + "it", + "y" + ], + [ + "Ġc", + "orn" + ], + [ + "Ġmo", + "ved" + ], + [ + "sh", + "ine" + ], + [ + "Ġ", + "3" + ], + [ + "Ġth", + "ough" + ], + [ + "Ġc", + "ir" + ], + [ + "um", + "b" + ], + [ + "Ġta", + "king" + ], + [ + "t", + "s" + ], + [ + "ĠT", + "weet" + ], + [ + "ap", + "e" + ], + [ + "Ġrain", + "bow" + ], + [ + "Ġthr", + "ow" + ], + [ + "Ġst", + "ar" + ], + [ + "Ġben", + "ch" + ], + [ + "Ġne", + "ck" + ], + [ + "ock", + "ed" + ], + [ + "Ġcreat", + "ure" + ], + [ + "Ġbu", + "bble" + ], + [ + "Ġl", + "ate" + ], + [ + "Ġadventure", + "s" + ], + [ + "Ġo", + "wl" + ], + [ + "ion", + "s" + ], + [ + "A", + "nd" + ], + [ + "Ġw", + "he" + ], + [ + "Ġp", + "ar" + ], + [ + "Ġshe", + "ll" + ], + [ + "Ġk", + "ite" + ], + [ + "ĠFl", + "uffy" + ], + [ + "in", + "ing" + ], + [ + "Ġm", + "il" + ], + [ + "u", + "nt" + ], + [ + "Ġf", + "ence" + ], + [ + "Ġm", + "ix" + ], + [ + "Ġli", + "ft" + ], + [ + "Ġaccident", + "ally" + ], + [ + "Ġw", + "ise" + ], + [ + "Ġhe", + "llo" + ], + [ + "Ġp", + "urple" + ], + [ + "pp", + "er" + ], + [ + "Ġon", + "to" + ], + [ + "Ġsp", + "in" + ], + [ + "t", + "en" + ], + [ + "our", + "s" + ], + [ + "Ġs", + "itting" + ], + [ + "id", + "ge" + ], + [ + "Ġsun", + "shine" + ], + [ + "Ġcut", + "e" + ], + [ + "Ġm", + "at" + ], + [ + "w", + "ard" + ], + [ + "Ġsh", + "op" + ], + [ + "Ġwh", + "ist" + ], + [ + "Ġd", + "ist" + ], + [ + "Ġor", + "ange" + ], + [ + "L", + "ila" + ], + [ + "Ġn", + "aught" + ], + [ + "ĠSam", + "my" + ], + [ + "Ġnaught", + "y" + ], + [ + "Ġn", + "est" + ], + [ + "o", + "se" + ], + [ + "or", + "s" + ], + [ + "ll", + "a" + ], + [ + "Ġmil", + "k" + ], + [ + "f", + "ish" + ], + [ + "Ġa", + "f" + ], + [ + "Ġb", + "oun" + ], + [ + "Ġp", + "hone" + ], + [ + "Ġarm", + "s" + ], + [ + "Ġe", + "as" + ], + [ + "n", + "ess" + ], + [ + "Ġcar", + "ry" + ], + [ + "ĠGra", + "ndma" + ], + [ + "o", + "g" + ], + [ + "Ġb", + "ought" + ], + [ + "Ġdri", + "ver" + ], + [ + "Ġinst", + "ead" + ], + [ + "L", + "ater" + ], + [ + "Ġtal", + "ked" + ], + [ + "Ġsear", + "ched" + ], + [ + "g", + "ged" + ], + [ + "Ġp", + "ig" + ], + [ + "Ġco", + "zy" + ], + [ + "Ġsun", + "ny" + ], + [ + "itt", + "y" + ], + [ + "Ġfeel", + "s" + ], + [ + "Ġf", + "an" + ], + [ + "Ġwonder", + "ed" + ], + [ + "Ġcom", + "es" + ], + [ + "Ġswim", + "ming" + ], + [ + "u", + "ched" + ], + [ + "Ġto", + "uched" + ], + [ + "Ġp", + "op" + ], + [ + "ĠIn", + "side" + ], + [ + "a", + "le" + ], + [ + "Ġl", + "ucky" + ], + [ + "Ġlaug", + "hing" + ], + [ + "Ġbel", + "ong" + ], + [ + "Ġsound", + "s" + ], + [ + "Ġw", + "om" + ], + [ + "Ġc", + "ave" + ], + [ + "le", + "t" + ], + [ + "Ġjourn", + "ey" + ], + [ + "Ġwom", + "an" + ], + [ + "Ġc", + "ou" + ], + [ + "Ġnot", + "hing" + ], + [ + "Ġco", + "at" + ], + [ + "Ġrelie", + "ved" + ], + [ + "Ġ", + "O" + ], + [ + "Ġpe", + "ace" + ], + [ + "ad", + "ow" + ], + [ + "Ġno", + "se" + ], + [ + "Ġbu", + "sy" + ], + [ + "B", + "ob" + ], + [ + "Ġp", + "ast" + ], + [ + "Ġapp", + "les" + ], + [ + "Ġs", + "ick" + ], + [ + "Ġr", + "ope" + ], + [ + "Ġpat", + "ient" + ], + [ + "Ġa", + "ct" + ], + [ + "Ġp", + "udd" + ], + [ + "Ġin", + "c" + ], + [ + "ra", + "id" + ], + [ + "Ġwatch", + "ing" + ], + [ + "Ġbran", + "ch" + ], + [ + "Ġaf", + "raid" + ], + [ + "Ġto", + "m" + ], + [ + "id", + "er" + ], + [ + "Ġday", + "s" + ], + [ + "Ġre", + "t" + ], + [ + "c", + "ing" + ], + [ + "ĠM", + "ary" + ], + [ + "Ġb", + "and" + ], + [ + "Ġfar", + "m" + ], + [ + "Ġhug", + "e" + ], + [ + "ĠTh", + "ank" + ], + [ + "at", + "or" + ], + [ + "Ġexc", + "iting" + ], + [ + "Ġdan", + "ced" + ], + [ + "th", + "day" + ], + [ + "Ġwa", + "r" + ], + [ + "ĠSo", + "on" + ], + [ + "b", + "all" + ], + [ + "Ġg", + "igg" + ], + [ + "ble", + "m" + ], + [ + "Ġz", + "oom" + ], + [ + "Ġm", + "ad" + ], + [ + "Ġv", + "ill" + ], + [ + "Ġfore", + "ver" + ], + [ + "Ġpro", + "blem" + ], + [ + "Ġcoll", + "ect" + ], + [ + "p", + "ed" + ], + [ + "ir", + "t" + ], + [ + "Ġmin", + "ut" + ], + [ + "Ġw", + "ra" + ], + [ + "h", + "o" + ], + [ + "Ġli", + "fe" + ], + [ + "Ġkne", + "e" + ], + [ + "c", + "es" + ], + [ + "D", + "o" + ], + [ + "c", + "er" + ], + [ + "m", + "p" + ], + [ + "ar", + "p" + ], + [ + "Ġk", + "ing" + ], + [ + "Ġas", + "leep" + ], + [ + "Ġret", + "urn" + ], + [ + "Ġsh", + "arp" + ], + [ + "Ġre", + "l" + ], + [ + "Ġp", + "ower" + ], + [ + "et", + "h" + ], + [ + "Ġhold", + "ing" + ], + [ + "Ġf", + "ing" + ], + [ + "Ġheal", + "thy" + ], + [ + "ĠM", + "ittens" + ], + [ + "Ġpr", + "ot" + ], + [ + "Ġme", + "et" + ], + [ + "Ġor", + "gan" + ], + [ + "Ġcle", + "ver" + ], + [ + "Ġspot", + "ted" + ], + [ + "Ġl", + "etter" + ], + [ + "Ġg", + "ather" + ], + [ + "al", + "m" + ], + [ + "O", + "f" + ], + [ + "e", + "re" + ], + [ + "Ġr", + "ound" + ], + [ + "Ġstor", + "m" + ], + [ + "Ġprot", + "ect" + ], + [ + "Ġgif", + "t" + ], + [ + "am", + "ed" + ], + [ + "Ġma", + "il" + ], + [ + "ĠJ", + "en" + ], + [ + "Ġbir", + "thday" + ], + [ + "Ġpre", + "s" + ], + [ + "Ġsa", + "f" + ], + [ + "Ġneighb", + "or" + ], + [ + "ep", + "end" + ], + [ + "Ġta", + "kes" + ], + [ + "s", + "c" + ], + [ + "Ġe", + "ar" + ], + [ + "Ġcom", + "fort" + ], + [ + "Ġ", + "ent" + ], + [ + "Ġre", + "p" + ], + [ + "Ġpers", + "on" + ], + [ + "Ġsmil", + "ing" + ], + [ + "ĠW", + "ith" + ], + [ + "Ġgent", + "ly" + ], + [ + "Ġpoin", + "ted" + ], + [ + "E", + "very" + ], + [ + "ow", + "ed" + ], + [ + "Ġsp", + "o" + ], + [ + "co", + "l" + ], + [ + "Ġtast", + "y" + ], + [ + "ul", + "ar" + ], + [ + "ĠJim", + "my" + ], + [ + "Ġpl", + "ane" + ], + [ + "st", + "ing" + ], + [ + "h", + "in" + ], + [ + "Ġh", + "oney" + ], + [ + "it", + "ch" + ], + [ + "Ġp", + "ill" + ], + [ + "Ġp", + "ass" + ], + [ + "itt", + "en" + ], + [ + "Ġm", + "atter" + ], + [ + "Ġad", + "m" + ], + [ + "Ġtast", + "ed" + ], + [ + "Ġsmell", + "ed" + ], + [ + "n", + "ic" + ], + [ + "o", + "ve" + ], + [ + "Ġe", + "ag" + ], + [ + "Ġsh", + "ining" + ], + [ + "He", + "y" + ], + [ + "Ġhop", + "e" + ], + [ + "Ġplac", + "es" + ], + [ + "Ġk", + "ick" + ], + [ + "ĠC", + "h" + ], + [ + "Ġpic", + "nic" + ], + [ + "Ġbott", + "le" + ], + [ + "Ġresp", + "ect" + ], + [ + "Ġb", + "lew" + ], + [ + "Ġapp", + "eared" + ], + [ + "Ġsu", + "pp" + ], + [ + "Tom", + "my" + ], + [ + "ĠC", + "ome" + ], + [ + "Ġsp", + "ider" + ], + [ + "Ġint", + "ere" + ], + [ + "Ġche", + "er" + ], + [ + "bo", + "ard" + ], + [ + "Ġwe", + "aring" + ], + [ + "Ġpres", + "ent" + ], + [ + "er", + "t" + ], + [ + "Ġm", + "ist" + ], + [ + "iz", + "e" + ], + [ + "Ġtrou", + "ble" + ], + [ + "Ġtri", + "p" + ], + [ + "Ġp", + "ract" + ], + [ + "Ġk", + "iss" + ], + [ + "Ġqu", + "est" + ], + [ + "o", + "nd" + ], + [ + "Ġas", + "h" + ], + [ + "Ġlo", + "ves" + ], + [ + "Ġtight", + "ly" + ], + [ + "g", + "g" + ], + [ + "ar", + "ge" + ], + [ + "Ġs", + "ur" + ], + [ + "Ġsp", + "read" + ], + [ + "a", + "ur" + ], + [ + "Ġw", + "ide" + ], + [ + "on", + "es" + ], + [ + "um", + "ber" + ], + [ + "Ġfe", + "ather" + ], + [ + "Ġspl", + "as" + ], + [ + "on", + "t" + ], + [ + "nd", + "p" + ], + [ + "Ġl", + "ater" + ], + [ + "Ġwho", + "le" + ], + [ + "Ġany", + "one" + ], + [ + "Ġbre", + "ad" + ], + [ + "M", + "olly" + ], + [ + "Ġd", + "es" + ], + [ + "Ġre", + "c" + ], + [ + "ell", + "a" + ], + [ + "Ġro", + "ad" + ], + [ + "I", + "n" + ], + [ + "Ġo", + "ce" + ], + [ + "Ġgi", + "ant" + ], + [ + "Ġoce", + "an" + ], + [ + "el", + "s" + ], + [ + "Ġpu", + "zz" + ], + [ + "Ġcook", + "ie" + ], + [ + "Ġp", + "an" + ], + [ + "Ġp", + "ath" + ], + [ + "Åĵ", + "I" + ], + [ + "Ġcon", + "fused" + ], + [ + "Ġscream", + "ed" + ], + [ + "Ġcloud", + "s" + ], + [ + "e", + "b" + ], + [ + "Ġimp", + "ress" + ], + [ + "Ġfr", + "ont" + ], + [ + "ĠG", + "o" + ], + [ + "Ġse", + "at" + ], + [ + "le", + "x" + ], + [ + "Ġl", + "ay" + ], + [ + "Ġwh", + "ich" + ], + [ + "l", + "a" + ], + [ + "Ġth", + "in" + ], + [ + "Ġte", + "ach" + ], + [ + "er", + "ly" + ], + [ + "Ġd", + "i" + ], + [ + "Ġansw", + "er" + ], + [ + "ung", + "le" + ], + [ + "Ġa", + "p" + ], + [ + "ss", + "ed" + ], + [ + "h", + "ile" + ], + [ + "u", + "b" + ], + [ + "Ġl", + "arge" + ], + [ + "Ġj", + "ungle" + ], + [ + "Ġprom", + "ise" + ], + [ + "Ġs", + "er" + ], + [ + "Ġb", + "er" + ], + [ + "Ġw", + "ild" + ], + [ + "Ġb", + "ark" + ], + [ + "Ġm", + "ach" + ], + [ + "oo", + "se" + ], + [ + "Ġbutt", + "on" + ], + [ + "ndp", + "a" + ], + [ + "G", + "ood" + ], + [ + "Ġmy", + "ster" + ], + [ + "Ġsn", + "ake" + ], + [ + "Ġvill", + "age" + ], + [ + "Ġh", + "or" + ], + [ + "ĠEm", + "ily" + ], + [ + "C", + "ome" + ], + [ + "Ġtoo", + "l" + ], + [ + "Ġput", + "s" + ], + [ + "Ġla", + "st" + ], + [ + "Ġim", + "ag" + ], + [ + "ĠJ", + "ake" + ], + [ + "L", + "ucy" + ], + [ + "Ġt", + "id" + ], + [ + "Ġb", + "an" + ], + [ + "le", + "br" + ], + [ + "Ġe", + "ventually" + ], + [ + "Ġsl", + "id" + ], + [ + "Ġwa", + "ve" + ], + [ + "Ġli", + "ve" + ], + [ + "Ġch", + "ased" + ], + [ + "ĠEvery", + "where" + ], + [ + "Ġce", + "lebr" + ], + [ + "Ġcray", + "ons" + ], + [ + "J", + "ust" + ], + [ + "Ġsp", + "ent" + ], + [ + "Ġwr", + "ite" + ], + [ + "Ġgi", + "ves" + ], + [ + "un", + "k" + ], + [ + "if", + "e" + ], + [ + "Ġdec", + "or" + ], + [ + "Ġfo", + "ot" + ], + [ + "Ġdel", + "ight" + ], + [ + "Ġso", + "l" + ], + [ + "Ġsand", + "w" + ], + [ + "Ġdir", + "t" + ], + [ + "ĠJ", + "enny" + ], + [ + "Ġtal", + "king" + ], + [ + "Ġtreat", + "s" + ], + [ + "Ġc", + "art" + ], + [ + "ch", + "ing" + ], + [ + "Ġwait", + "ing" + ], + [ + "Ġhid", + "ing" + ], + [ + "Ġsnow", + "man" + ], + [ + "Ġintere", + "sting" + ], + [ + "Ġh", + "on" + ], + [ + "on", + "y" + ], + [ + "Ġshe", + "lf" + ], + [ + "Ġtast", + "e" + ], + [ + "e", + "en" + ], + [ + "J", + "im" + ], + [ + "Ġst", + "ep" + ], + [ + "M", + "um" + ], + [ + "Ġhelp", + "ful" + ], + [ + "Ġfe", + "et" + ], + [ + "Ġsh", + "y" + ], + [ + "Ġgo", + "es" + ], + [ + "Ġv", + "al" + ], + [ + "Ġe", + "mb" + ], + [ + "Ġst", + "ood" + ], + [ + "Ġsh", + "r" + ], + [ + "Ġbuild", + "ing" + ], + [ + "Ġo", + "ven" + ], + [ + "Ġst", + "ret" + ], + [ + "Ġlove", + "ly" + ], + [ + "Ġduck", + "s" + ], + [ + "Ġcorn", + "er" + ], + [ + "Ġwas", + "h" + ], + [ + "ck", + "s" + ], + [ + "Ġsp", + "e" + ], + [ + "Ġpretend", + "ed" + ], + [ + "Ġroll", + "ed" + ], + [ + "Ġru", + "de" + ], + [ + "o", + "ld" + ], + [ + "Ġc", + "alm" + ], + [ + "Ġp", + "ile" + ], + [ + "Ġte", + "eth" + ], + [ + "Ġjump", + "ing" + ], + [ + "Ġz", + "oo" + ], + [ + "Ġpudd", + "le" + ], + [ + "Ġtid", + "y" + ], + [ + "M", + "ama" + ], + [ + "ig", + "er" + ], + [ + "ip", + "e" + ], + [ + "Ġbug", + "s" + ], + [ + "Ġash", + "amed" + ], + [ + "Ġb", + "at" + ], + [ + "in", + "a" + ], + [ + "ĠR", + "e" + ], + [ + "Ġo", + "b" + ], + [ + "Ġst", + "ra" + ], + [ + "ĠS", + "t" + ], + [ + "Ġwhist", + "le" + ], + [ + "Ġban", + "an" + ], + [ + "n", + "ot" + ], + [ + "Ġn", + "et" + ], + [ + "Ġforg", + "ive" + ], + [ + "Ġprin", + "ce" + ], + [ + "en", + "er" + ], + [ + "Ġsh", + "irt" + ], + [ + "Ġs", + "el" + ], + [ + "Ġc", + "oo" + ], + [ + "Ġemb", + "ar" + ], + [ + "ump", + "y" + ], + [ + "Ġspark", + "ly" + ], + [ + "Ġrem", + "ind" + ], + [ + "Ġs", + "ug" + ], + [ + "Ġc", + "ross" + ], + [ + "Ġm", + "ind" + ], + [ + "Ġcan", + "not" + ], + [ + "Ġsuc", + "cess" + ], + [ + "Ġembar", + "ra" + ], + [ + "S", + "oon" + ], + [ + "Ġm", + "ot" + ], + [ + "Ġsh", + "aring" + ], + [ + "Ġwor", + "m" + ], + [ + "Ġfo", + "ld" + ], + [ + "Ġpeace", + "ful" + ], + [ + "Ġw", + "ore" + ], + [ + "Ġm", + "ountain" + ], + [ + "Ġbl", + "ow" + ], + [ + "l", + "ice" + ], + [ + "Ġ", + "ign" + ], + [ + "Ġbe", + "ll" + ], + [ + "Ġfur", + "ry" + ], + [ + "Ġgra", + "b" + ], + [ + "al", + "s" + ], + [ + "omet", + "imes" + ], + [ + "Ġwhen", + "ever" + ], + [ + "Ġp", + "our" + ], + [ + "v", + "ous" + ], + [ + "ll", + "ie" + ], + [ + "Ġpu", + "sh" + ], + [ + "Ġbre", + "ath" + ], + [ + "Ġmach", + "ine" + ], + [ + "Ġsug", + "ar" + ], + [ + "Ġch", + "ick" + ], + [ + "Ġcreat", + "ive" + ], + [ + "S", + "t" + ], + [ + "n", + "ed" + ], + [ + "Ġb", + "al" + ], + [ + "Ġst", + "ir" + ], + [ + "Ġdo", + "lp" + ], + [ + "Ġtri", + "es" + ], + [ + "N", + "ow" + ], + [ + "Ġl", + "ad" + ], + [ + "ĠL", + "e" + ], + [ + "Ġm", + "ed" + ], + [ + "Ġp", + "ool" + ], + [ + "Ġg", + "ener" + ], + [ + "Ġsa", + "l" + ], + [ + "or", + "k" + ], + [ + "ul", + "t" + ], + [ + "Ġspr", + "ay" + ], + [ + "Ġsh", + "ip" + ], + [ + "i", + "an" + ], + [ + "Ġh", + "um" + ], + [ + "Ġhe", + "l" + ], + [ + "is", + "a" + ], + [ + "Ġsel", + "fish" + ], + [ + "Ġh", + "ar" + ], + [ + "Ġfa", + "ces" + ], + [ + "Ġmess", + "y" + ], + [ + "Ġ", + "'" + ], + [ + "Ġp", + "le" + ], + [ + "ĠS", + "ome" + ], + [ + "ĠH", + "ow" + ], + [ + "Ġgo", + "ld" + ], + [ + "Ġcho", + "col" + ], + [ + "P", + "lease" + ], + [ + "Ġminut", + "es" + ], + [ + "Ġpill", + "ow" + ], + [ + "M", + "y" + ], + [ + "a", + "king" + ], + [ + "Ġt", + "un" + ], + [ + "Ġm", + "issed" + ], + [ + "ig", + "hed" + ], + [ + "Ġv", + "an" + ], + [ + "Ġfix", + "ed" + ], + [ + "Ġfight", + "ing" + ], + [ + "o", + "in" + ], + [ + "Ġf", + "ig" + ], + [ + "Ġembarra", + "ssed" + ], + [ + "Ġme", + "ant" + ], + [ + "Ġdri", + "ve" + ], + [ + "Ġtom", + "orrow" + ], + [ + "Ġh", + "ours" + ], + [ + "Ġc", + "a" + ], + [ + "Ġn", + "er" + ], + [ + "Ġsp", + "icy" + ], + [ + "Ġknow", + "ing" + ], + [ + "ĠP", + "lease" + ], + [ + "Sara", + "h" + ], + [ + "Ġlad", + "der" + ], + [ + "it", + "al" + ], + [ + "Ġhor", + "se" + ], + [ + "at", + "o" + ], + [ + "Ġu", + "g" + ], + [ + "ĠJ", + "ust" + ], + [ + "Ġtr", + "ust" + ], + [ + "Ġho", + "sp" + ], + [ + "Ġug", + "ly" + ], + [ + "Ġhosp", + "ital" + ], + [ + "Ġche", + "w" + ], + [ + "Ġbed", + "room" + ], + [ + "Ġeas", + "y" + ], + [ + "Ġner", + "vous" + ], + [ + "Ġro", + "b" + ], + [ + "Ġthank", + "ful" + ], + [ + "Ġcarrot", + "s" + ], + [ + "D", + "ad" + ], + [ + "ell", + "y" + ], + [ + "Ġgre", + "w" + ], + [ + "Ġcol", + "our" + ], + [ + "Ġbar", + "ked" + ], + [ + "Ġcou", + "ch" + ], + [ + "Ġn", + "ut" + ], + [ + "Ġme", + "adow" + ], + [ + "Ġta", + "ught" + ], + [ + "Ġunderst", + "ood" + ], + [ + "Ġdisapp", + "oin" + ], + [ + "Ġp", + "as" + ], + [ + "an", + "o" + ], + [ + "Ġof", + "ten" + ], + [ + "Ġcl", + "ass" + ], + [ + "Ġpas", + "sed" + ], + [ + "Ġkn", + "ocked" + ], + [ + "Ġland", + "ed" + ], + [ + "Ġm", + "ic" + ], + [ + "ir", + "d" + ], + [ + "Ġfe", + "ar" + ], + [ + "Ġco", + "in" + ], + [ + "Ġmu", + "d" + ], + [ + "w", + "ork" + ], + [ + "Ġd", + "eter" + ], + [ + "Ġre", + "g" + ], + [ + "Ġreturn", + "ed" + ], + [ + "Ġdeter", + "m" + ], + [ + "Ġh", + "ur" + ], + [ + "Ġfin", + "ish" + ], + [ + "iz", + "z" + ], + [ + "Ġte", + "a" + ], + [ + "Ġsmo", + "oth" + ], + [ + "m", + "o" + ], + [ + "Ġst", + "uff" + ], + [ + "Ġmo", + "v" + ], + [ + "Ġgener", + "ous" + ], + [ + "T", + "o" + ], + [ + "Ġ", + "On" + ], + [ + "is", + "p" + ], + [ + "Ġdr", + "um" + ], + [ + "o", + "bby" + ], + [ + "Ġbubble", + "s" + ], + [ + "Ġrob", + "ot" + ], + [ + "Ġthe", + "se" + ], + [ + "Ġse", + "ed" + ], + [ + "Ġgl", + "ass" + ], + [ + "Ġdisappoin", + "ted" + ], + [ + "g", + "round" + ], + [ + "Ġr", + "ing" + ], + [ + "Ġc", + "ard" + ], + [ + "Ġhe", + "ars" + ], + [ + "Ġcar", + "ried" + ], + [ + "Ġmo", + "on" + ], + [ + "Ġchocol", + "ate" + ], + [ + "Ġlea", + "f" + ], + [ + "Ġsnack", + "s" + ], + [ + "Ġcir", + "c" + ], + [ + "Ġfan", + "cy" + ], + [ + "d", + "le" + ], + [ + "Ġsp", + "end" + ], + [ + "ĠA", + "nn" + ], + [ + "ust", + "r" + ], + [ + "Ġcra", + "b" + ], + [ + "Ġsong", + "s" + ], + [ + "p", + "ack" + ], + [ + "Ġp", + "ir" + ], + [ + "eci", + "ally" + ], + [ + "os", + "aur" + ], + [ + "Ġso", + "ld" + ], + [ + "Ġbr", + "idge" + ], + [ + "ĠE", + "ven" + ], + [ + "Ġsleep", + "y" + ], + [ + "Ġhon", + "est" + ], + [ + "Ġm", + "ir" + ], + [ + "ĠB", + "r" + ], + [ + "Ġpa", + "w" + ], + [ + "p", + "ecially" + ], + [ + "Ġa", + "v" + ], + [ + "ĠW", + "hy" + ], + [ + "iz", + "zy" + ], + [ + "Ġche", + "st" + ], + [ + "Ġwo", + "lf" + ], + [ + "Ġdin", + "osaur" + ], + [ + "Ġes", + "pecially" + ], + [ + "Ġm", + "ap" + ], + [ + "Ġg", + "as" + ], + [ + "or", + "y" + ], + [ + "Ġch", + "ange" + ], + [ + "Ġgr", + "oup" + ], + [ + "Ġgather", + "ed" + ], + [ + "Ġs", + "our" + ], + [ + "Ġpo", + "or" + ], + [ + "Ġspl", + "ash" + ], + [ + "Ġwa", + "ves" + ], + [ + "Ġfor", + "ward" + ], + [ + "Ġcl", + "own" + ], + [ + "Ġmyster", + "ious" + ], + [ + "r", + "or" + ], + [ + "Ġbo", + "dy" + ], + [ + "Ġfr", + "ustr" + ], + [ + "Ġcra", + "w" + ], + [ + "Ġf", + "ake" + ], + [ + "Ġp", + "in" + ], + [ + "Ġo", + "k" + ], + [ + "Ġr", + "ad" + ], + [ + "Ġwh", + "isp" + ], + [ + "Ġtw", + "irl" + ], + [ + "Ġstr", + "ing" + ], + [ + "Ġsmell", + "y" + ], + [ + "ĠTo", + "gether" + ], + [ + "s", + "ist" + ], + [ + "Ġc", + "and" + ], + [ + "Ġg", + "ate" + ], + [ + "Ġsc", + "ar" + ], + [ + "ĠK", + "itty" + ], + [ + "Ġneck", + "l" + ], + [ + "B", + "illy" + ], + [ + "Ġmot", + "or" + ], + [ + "y", + "ing" + ], + [ + "ot", + "e" + ], + [ + "Ġsh", + "ore" + ], + [ + "ur", + "se" + ], + [ + "Ġje", + "w" + ], + [ + "Ġwe", + "ak" + ], + [ + "Ġgr", + "umpy" + ], + [ + "Ġfing", + "er" + ], + [ + "Ġsa", + "ck" + ], + [ + "Ġk", + "itten" + ], + [ + "Ġpol", + "ite" + ], + [ + "Ġgl", + "ue" + ], + [ + "u", + "ce" + ], + [ + "Ġn", + "umber" + ], + [ + "air", + "s" + ], + [ + "Ġes", + "c" + ], + [ + "Ġmir", + "ror" + ], + [ + "Ġshe", + "ep" + ], + [ + "Ġv", + "ase" + ], + [ + "Ġplant", + "s" + ], + [ + "Ġber", + "ries" + ], + [ + "mo", + "st" + ], + [ + "B", + "e" + ], + [ + "d", + "en" + ], + [ + "Ġj", + "elly" + ], + [ + "ĠThe", + "ir" + ], + [ + "Ġe", + "mp" + ], + [ + "itt", + "er" + ], + [ + "The", + "re" + ], + [ + "Ġsc", + "r" + ], + [ + "Ġgr", + "ay" + ], + [ + "Ġpuzz", + "le" + ], + [ + "Ġneckl", + "ace" + ], + [ + "Ġm", + "aybe" + ], + [ + "Ġpl", + "ate" + ], + [ + "Ġal", + "most" + ], + [ + "us", + "ie" + ], + [ + "Ġfrustr", + "ated" + ], + [ + "Ġemp", + "ty" + ], + [ + "p", + "ort" + ], + [ + "or", + "ry" + ], + [ + "Ġth", + "ick" + ], + [ + "Ġj", + "am" + ], + [ + "Ġsk", + "ipped" + ], + [ + "ens", + "ive" + ], + [ + "ĠT", + "eddy" + ], + [ + "ag", + "ed" + ], + [ + "Ġclos", + "et" + ], + [ + "Ġhid", + "den" + ], + [ + "Ġexp", + "ensive" + ], + [ + "Ġdeterm", + "ined" + ], + [ + "Ġb", + "ored" + ], + [ + "Ġl", + "it" + ], + [ + "Ġd", + "rew" + ], + [ + "Ġle", + "gs" + ], + [ + "Ġho", + "pping" + ], + [ + "at", + "es" + ], + [ + "ĠS", + "ometimes" + ], + [ + "Ġstick", + "s" + ], + [ + "Ġon", + "es" + ], + [ + "ĠA", + "t" + ], + [ + "Ġbird", + "ie" + ], + [ + "ĠD", + "on" + ], + [ + "Ġpol", + "ice" + ], + [ + "Ġfruit", + "s" + ], + [ + "Ġt", + "ick" + ], + [ + "Ġa", + "unt" + ], + [ + "Ġw", + "is" + ], + [ + "Ġb", + "ake" + ], + [ + "Ġl", + "ine" + ], + [ + "ot", + "s" + ], + [ + "Ġst", + "one" + ], + [ + "Ġdog", + "s" + ], + [ + "au", + "l" + ], + [ + "Ġfig", + "ure" + ], + [ + "J", + "ane" + ], + [ + "Ġs", + "ight" + ], + [ + "Ġc", + "ries" + ], + [ + "ĠB", + "e" + ], + [ + "Ġmo", + "ving" + ], + [ + "Ġcont", + "ent" + ], + [ + "ĠBr", + "own" + ], + [ + "Ġsa", + "ved" + ], + [ + "Ġcl", + "oth" + ], + [ + "?\"", + "." + ], + [ + "bo", + "x" + ], + [ + "Ġbran", + "ches" + ], + [ + "Ġpack", + "ed" + ], + [ + "Ġpower", + "ful" + ], + [ + "Ġsplas", + "hed" + ], + [ + "Ġfe", + "ed" + ], + [ + "est", + "ed" + ], + [ + "Ġyell", + "ed" + ], + [ + "W", + "here" + ], + [ + "d", + "ge" + ], + [ + "k", + "in" + ], + [ + "Ġt", + "iger" + ], + [ + "Ġp", + "ay" + ], + [ + "id", + "dle" + ], + [ + "oo", + "p" + ], + [ + "ĠB", + "ella" + ], + [ + "Ġex", + "am" + ], + [ + "Ġta", + "ken" + ], + [ + "Ġcr", + "ane" + ], + [ + "Ġspo", + "il" + ], + [ + "'", + "d" + ], + [ + "Ġha", + "ng" + ], + [ + "Ġfl", + "ag" + ], + [ + "Ġansw", + "ered" + ], + [ + "Ġdisco", + "vered" + ], + [ + "ĠLe", + "o" + ], + [ + "u", + "sh" + ], + [ + "Ġsa", + "ve" + ], + [ + "Ġsee", + "k" + ], + [ + "Ġexc", + "ite" + ], + [ + "Ġpr", + "ay" + ], + [ + "Ġjo", + "g" + ], + [ + "Ġstand", + "ing" + ], + [ + "Ġdolp", + "hin" + ], + [ + "Ġj", + "ack" + ], + [ + "A", + "re" + ], + [ + "er", + "a" + ], + [ + "ra", + "g" + ], + [ + "Ġfl", + "ut" + ], + [ + "get", + "able" + ], + [ + "Ġb", + "oring" + ], + [ + "Ġlo", + "ck" + ], + [ + "Ġinc", + "red" + ], + [ + "Ġexcite", + "ment" + ], + [ + "Ġs", + "ighed" + ], + [ + "Ġw", + "ip" + ], + [ + "Ġf", + "is" + ], + [ + "Ġf", + "ill" + ], + [ + "Ġso", + "ap" + ], + [ + "ĠM", + "um" + ], + [ + "ol", + "a" + ], + [ + "Ġple", + "ased" + ], + [ + "Ġjack", + "et" + ], + [ + "l", + "ight" + ], + [ + "Ġs", + "ugg" + ], + [ + "Ġm", + "iddle" + ], + [ + "Ġe", + "ld" + ], + [ + "Ġwr", + "ote" + ], + [ + "Ġballoon", + "s" + ], + [ + "Ġorgan", + "ized" + ], + [ + "qu", + "e" + ], + [ + "Ġsail", + "ed" + ], + [ + "The", + "ir" + ], + [ + "Ġtell", + "s" + ], + [ + "Ġo", + "y" + ], + [ + "Ġst", + "ones" + ], + [ + "Ġbo", + "ard" + ], + [ + "Ġsp", + "un" + ], + [ + "Ġve", + "getable" + ], + [ + "l", + "ies" + ], + [ + "Ġ", + "ind" + ], + [ + "in", + "ess" + ], + [ + "Ġwor", + "king" + ], + [ + "l", + "ower" + ], + [ + "al", + "k" + ], + [ + "if", + "f" + ], + [ + "Ġpar", + "rot" + ], + [ + "St", + "op" + ], + [ + "Ġscar", + "f" + ], + [ + "Ġwa", + "gged" + ], + [ + "Ġcl", + "ap" + ], + [ + "Ġme", + "asure" + ], + [ + "as", + "ing" + ], + [ + "Ġte", + "ars" + ], + [ + "bo", + "dy" + ], + [ + "Ġcomfort", + "able" + ], + [ + "Ġle", + "g" + ], + [ + "Ġsw", + "an" + ], + [ + "Ġtr", + "ue" + ], + [ + "Ġfr", + "ight" + ], + [ + "Ġdr", + "op" + ], + [ + "Ġsmo", + "ke" + ], + [ + "Ġfeather", + "s" + ], + [ + "Ġincred", + "ible" + ], + [ + "g", + "gs" + ], + [ + "u", + "al" + ], + [ + "Ġt", + "ie" + ], + [ + "Ġc", + "ap" + ], + [ + "Ġlo", + "se" + ], + [ + "Ġpu", + "s" + ], + [ + "Ġeag", + "er" + ], + [ + "Ġmov", + "ie" + ], + [ + "f", + "ic" + ], + [ + "Ġc", + "op" + ], + [ + "Ġe", + "ggs" + ], + [ + "oo", + "f" + ], + [ + "Ġr", + "id" + ], + [ + "Ġco", + "vered" + ], + [ + "Ġf", + "il" + ], + [ + "Ġle", + "m" + ], + [ + "Ġgra", + "p" + ], + [ + "Ġdisapp", + "eared" + ], + [ + "Ġrec", + "ord" + ], + [ + "ĠT", + "V" + ], + [ + "ĠB", + "l" + ], + [ + "Ġex", + "t" + ], + [ + "Ġor", + "d" + ], + [ + "Ġgigg", + "led" + ], + [ + "c", + "orn" + ], + [ + "Ġp", + "ri" + ], + [ + "Ġha", + "m" + ], + [ + "Ġr", + "are" + ], + [ + "Ġyour", + "self" + ], + [ + "Ġen", + "g" + ], + [ + "Ġwh", + "ale" + ], + [ + "Ġkn", + "ife" + ], + [ + "ĠL", + "isa" + ], + [ + "Ġte", + "am" + ], + [ + "ĠJohn", + "ny" + ], + [ + "z", + "ed" + ], + [ + "Ġt", + "est" + ], + [ + "Ġb", + "one" + ], + [ + "Ġd", + "izzy" + ], + [ + "Ġli", + "br" + ], + [ + "Ġwra", + "p" + ], + [ + "Ġquest", + "ions" + ], + [ + "i", + "v" + ], + [ + "Ġl", + "ying" + ], + [ + "Ġst", + "amp" + ], + [ + "Ġli", + "cked" + ], + [ + "Ġno", + "isy" + ], + [ + "Ġkind", + "ness" + ], + [ + "Ġdif", + "fic" + ], + [ + "Ġdraw", + "ing" + ], + [ + "app", + "ing" + ], + [ + "Jim", + "my" + ], + [ + "Ġstra", + "w" + ], + [ + "Ġb", + "urn" + ], + [ + "Ġwa", + "gon" + ], + [ + "at", + "ely" + ], + [ + "Ġgr", + "own" + ], + [ + "Ġrock", + "et" + ], + [ + "Ġpop", + "corn" + ], + [ + "S", + "ally" + ], + [ + "Ġb", + "atter" + ], + [ + "Ġwa", + "nd" + ], + [ + "Ġan", + "c" + ], + [ + "ĠE", + "nd" + ], + [ + "Ġsw", + "ord" + ], + [ + "Ġhappen", + "ing" + ], + [ + "Ġlift", + "ed" + ], + [ + "Ġbal", + "ance" + ], + [ + "Ġext", + "ra" + ], + [ + "S", + "orry" + ], + [ + "h", + "a" + ], + [ + "Ġwa", + "nder" + ], + [ + "Ġf", + "at" + ], + [ + "ĠM", + "ark" + ], + [ + "ĠE", + "lla" + ], + [ + "Ġbec", + "ome" + ], + [ + "Ġtr", + "ack" + ], + [ + "Ġsmall", + "er" + ], + [ + "Ġmist", + "ake" + ], + [ + "Ġto", + "ast" + ], + [ + "at", + "ing" + ], + [ + "Ġman", + "aged" + ], + [ + "sel", + "ves" + ], + [ + "ĠP", + "eter" + ], + [ + "Ġbath", + "room" + ], + [ + "Ġhar", + "m" + ], + [ + "Ġdiffic", + "ult" + ], + [ + "Ġtw", + "ist" + ], + [ + "Ġoff", + "ered" + ], + [ + "Ġru", + "ined" + ], + [ + "Ġmat", + "ch" + ], + [ + "Ġanc", + "ient" + ], + [ + "Ġg", + "ar" + ], + [ + "ill", + "i" + ], + [ + "Ġdist", + "ance" + ], + [ + "Ġplay", + "ground" + ], + [ + "Ġla", + "zy" + ], + [ + "Ġdan", + "cing" + ], + [ + "Ġadvent", + "ur" + ], + [ + "a", + "f" + ], + [ + "ĠB", + "a" + ], + [ + "Ġcall", + "ing" + ], + [ + "Ġclean", + "ed" + ], + [ + "Ġcr", + "own" + ], + [ + "orm", + "al" + ], + [ + "Ġon", + "ce" + ], + [ + "Ġbo", + "ss" + ], + [ + "Ġse", + "ll" + ], + [ + "Ġwalk", + "s" + ], + [ + "Ġfright", + "ened" + ], + [ + "er", + "ri" + ], + [ + "Ġm", + "ask" + ], + [ + "Ġfl", + "uffy" + ], + [ + "em", + "o" + ], + [ + "Ġ", + "Z" + ], + [ + "Ġf", + "rag" + ], + [ + "Ġd", + "est" + ], + [ + "Ġn", + "ormal" + ], + [ + "Ġy", + "og" + ], + [ + "Ġpick", + "s" + ], + [ + "Ġey", + "e" + ], + [ + "Ġdr", + "ank" + ], + [ + "Ġcir", + "cle" + ], + [ + "Ġadventur", + "ous" + ], + [ + "S", + "pot" + ], + [ + "Ġt", + "ent" + ], + [ + "Ġp", + "ress" + ], + [ + "Ġk", + "issed" + ], + [ + "ic", + "t" + ], + [ + "Ġsc", + "iss" + ], + [ + "Ġget", + "s" + ], + [ + "Ġtr", + "unk" + ], + [ + "Ġche", + "ck" + ], + [ + "Ġenjoy", + "ing" + ], + [ + "Ġshell", + "s" + ], + [ + "Ġeld", + "erly" + ], + [ + "Ġsciss", + "ors" + ], + [ + "Ġt", + "erri" + ], + [ + "Ġto", + "ug" + ], + [ + "ag", + "es" + ], + [ + "Ġste", + "al" + ], + [ + "M", + "e" + ], + [ + "re", + "ady" + ], + [ + "et", + "e" + ], + [ + "Ġal", + "ready" + ], + [ + "ĠL", + "ittle" + ], + [ + "Ġju", + "icy" + ], + [ + "Ġstu", + "dy" + ], + [ + "Ġh", + "orn" + ], + [ + "er", + "p" + ], + [ + "Ġp", + "izz" + ], + [ + "Ġall", + "ig" + ], + [ + "Ġloud", + "er" + ], + [ + "Ġtow", + "el" + ], + [ + "Ġcirc", + "les" + ], + [ + "Ġle", + "ad" + ], + [ + "Ġro", + "ar" + ], + [ + "Ġro", + "se" + ], + [ + "Ġsl", + "ipped" + ], + [ + "Ġte", + "le" + ], + [ + "Ġfam", + "ous" + ], + [ + "Ġar", + "g" + ], + [ + "Ġcheer", + "ful" + ], + [ + "ĠRe", + "x" + ], + [ + "Ġtoug", + "h" + ], + [ + "i", + "pper" + ], + [ + "Ġb", + "arn" + ], + [ + "Ġf", + "ool" + ], + [ + "Ġd", + "ig" + ], + [ + "Ġd", + "ish" + ], + [ + "Ġagain", + "st" + ], + [ + "Ġde", + "er" + ], + [ + "Ġsweet", + "ie" + ], + [ + "Ġeng", + "ine" + ], + [ + "Ġ", + "ed" + ], + [ + "Ġf", + "ather" + ], + [ + "Ġco", + "ver" + ], + [ + "Ġyour", + "s" + ], + [ + "Ġneed", + "s" + ], + [ + "Jo", + "e" + ], + [ + "Ġbag", + "s" + ], + [ + "Ġspin", + "ning" + ], + [ + "S", + "ue" + ], + [ + "Ġa", + "w" + ], + [ + "an", + "ged" + ], + [ + "am", + "ond" + ], + [ + "Ġy", + "arn" + ], + [ + "op", + "h" + ], + [ + "Ġgi", + "ving" + ], + [ + "Ġdi", + "amond" + ], + [ + "Ġsandw", + "ich" + ], + [ + "Ġind", + "epend" + ], + [ + "Ġfrag", + "ile" + ], + [ + "W", + "ho" + ], + [ + "ce", + "pt" + ], + [ + "Ġno", + "sy" + ], + [ + "Ġtw", + "ig" + ], + [ + "Ġhurt", + "s" + ], + [ + "Ġsugg", + "ested" + ], + [ + "Ġlem", + "on" + ], + [ + "i", + "est" + ], + [ + "in", + "ation" + ], + [ + "or", + "ge" + ], + [ + "Ġno", + "d" + ], + [ + "Ġbl", + "oom" + ], + [ + "Ġho", + "se" + ], + [ + "Ġbox", + "es" + ], + [ + "Ġreal", + "ised" + ], + [ + "Ġfr", + "own" + ], + [ + "ac", + "he" + ], + [ + "Ġrel", + "ax" + ], + [ + "Ġham", + "mer" + ], + [ + "Ġa", + "dded" + ], + [ + "Ġwe", + "ird" + ], + [ + "Ġmo", + "d" + ], + [ + "Ġcr", + "ack" + ], + [ + "Ġallig", + "ator" + ], + [ + "A", + "my" + ], + [ + "G", + "ra" + ], + [ + "o", + "ot" + ], + [ + "Ġsh", + "ake" + ], + [ + "ra", + "ge" + ], + [ + "Ġlearn", + "ing" + ], + [ + "Ġvegetable", + "s" + ], + [ + "Ġa", + "dd" + ], + [ + "Ġm", + "elt" + ], + [ + "ĠS", + "usie" + ], + [ + "el", + "p" + ], + [ + "ĠB", + "obby" + ], + [ + "Ġtra", + "p" + ], + [ + "Ġcoll", + "ar" + ], + [ + "Ġstu", + "bb" + ], + [ + "i", + "que" + ], + [ + "Ġb", + "orrow" + ], + [ + "Ġp", + "ump" + ], + [ + "Ġloud", + "ly" + ], + [ + "Ġsear", + "ch" + ], + [ + "Ġchick", + "en" + ], + [ + "Ġpizz", + "a" + ], + [ + "Ġindepend", + "ent" + ], + [ + "Ġstubb", + "orn" + ], + [ + "Ġre", + "f" + ], + [ + "Ġas", + "king" + ], + [ + "Ġbr", + "illi" + ], + [ + "Ġdec", + "ide" + ], + [ + "Ġfl", + "ash" + ], + [ + "Ġcat", + "erp" + ], + [ + "Ġqu", + "een" + ], + [ + "Ġsn", + "ee" + ], + [ + "ĠR", + "ose" + ], + [ + "Ġinv", + "ited" + ], + [ + "Ġstraw", + "ber" + ], + [ + "Ġfool", + "ish" + ], + [ + "Ġcaterp", + "ill" + ], + [ + "g", + "y" + ], + [ + "Ġt", + "ur" + ], + [ + "Ġd", + "epend" + ], + [ + "Ġg", + "uit" + ], + [ + "Ġst", + "ri" + ], + [ + "Ġkn", + "ight" + ], + [ + "Ġboy", + "s" + ], + [ + "Ġpe", + "e" + ], + [ + "Ġcom", + "b" + ], + [ + "Ġcle", + "ar" + ], + [ + "Ġun", + "ique" + ], + [ + "Ġbutt", + "ons" + ], + [ + "ĠTweet", + "y" + ], + [ + "Ġcolour", + "ful" + ], + [ + "Ġesc", + "ape" + ], + [ + "Ġbrilli", + "ant" + ], + [ + "Ġr", + "ich" + ], + [ + "ĠB", + "u" + ], + [ + "Ġsay", + "ing" + ], + [ + "Ġro", + "de" + ], + [ + "Ġsk", + "in" + ], + [ + "Ġgar", + "age" + ], + [ + "H", + "elp" + ], + [ + "d", + "om" + ], + [ + "Ġm", + "ummy" + ], + [ + "Ġbe", + "ak" + ], + [ + "ur", + "ed" + ], + [ + "a", + "w" + ], + [ + "e", + "orge" + ], + [ + "r", + "op" + ], + [ + "ed", + "ient" + ], + [ + "Ġun", + "us" + ], + [ + "Ġoy", + "ster" + ], + [ + "Ġtur", + "key" + ], + [ + "Ġ", + "OK" + ], + [ + "Ġf", + "ountain" + ], + [ + "Ġp", + "atter" + ], + [ + "Ġwe", + "ek" + ], + [ + "Ġch", + "ase" + ], + [ + "Ġru", + "bb" + ], + [ + "Ġcount", + "ed" + ], + [ + "Ġreg", + "ular" + ], + [ + "Ġwip", + "ed" + ], + [ + "Ġguit", + "ar" + ], + [ + "Ġunus", + "ual" + ], + [ + "l", + "and" + ], + [ + "Ġu", + "mb" + ], + [ + "Ġfor", + "k" + ], + [ + "Ġshow", + "s" + ], + [ + "Ġlight", + "s" + ], + [ + "Ġwall", + "s" + ], + [ + "Ġdream", + "ed" + ], + [ + "Ġbanan", + "a" + ], + [ + "Ġsuccess", + "ful" + ], + [ + "Ġjew", + "el" + ], + [ + "Ġpatter", + "n" + ], + [ + "Ġumb", + "re" + ], + [ + "c", + "il" + ], + [ + "r", + "ay" + ], + [ + "in", + "al" + ], + [ + "Ġl", + "ow" + ], + [ + "im", + "ed" + ], + [ + "Ġsw", + "e" + ], + [ + "Ġcry", + "st" + ], + [ + "Ġpen", + "cil" + ], + [ + "Ġsaf", + "ely" + ], + [ + "Ġtool", + "s" + ], + [ + "W", + "ell" + ], + [ + "Ġt", + "ape" + ], + [ + "Ġst", + "ream" + ], + [ + "ce", + "ed" + ], + [ + "Ġre", + "sc" + ], + [ + "Ġab", + "ove" + ], + [ + "Ġhard", + "er" + ], + [ + "Ġgu", + "ess" + ], + [ + "Ġbott", + "om" + ], + [ + "Ġmark", + "et" + ], + [ + "Ġimpress", + "ed" + ], + [ + "Ġwa", + "ff" + ], + [ + "an", + "ut" + ], + [ + "ig", + "inal" + ], + [ + "Ġnot", + "ice" + ], + [ + "ill", + "a" + ], + [ + "Ġno", + "ds" + ], + [ + "Ġpe", + "anut" + ], + [ + "ol", + "og" + ], + [ + "Ġpat", + "ch" + ], + [ + "Ġcraw", + "led" + ], + [ + "Ġcaterpill", + "ar" + ], + [ + "n", + "a" + ], + [ + "Ġt", + "ank" + ], + [ + "Ġt", + "ask" + ], + [ + "Ġt", + "ire" + ], + [ + "Ġg", + "un" + ], + [ + "Ġlo", + "ad" + ], + [ + "Ġcomp", + "et" + ], + [ + "umb", + "led" + ], + [ + "Ġap", + "olog" + ], + [ + "Ġsol", + "ve" + ], + [ + "H", + "ow" + ], + [ + "M", + "ummy" + ], + [ + "f", + "ast" + ], + [ + "h", + "ip" + ], + [ + "Ġ", + "No" + ], + [ + "Ġb", + "itter" + ], + [ + "ar", + "l" + ], + [ + "Ġe", + "ars" + ], + [ + "Ġsh", + "one" + ], + [ + "Ġbr", + "ush" + ], + [ + "ĠF", + "in" + ], + [ + "Ġor", + "iginal" + ], + [ + "Ġfi", + "er" + ], + [ + "Ġgl", + "ow" + ], + [ + "ĠN", + "emo" + ], + [ + "Ġbreak", + "fast" + ], + [ + "Ġadm", + "ired" + ], + [ + "ĠBl", + "ue" + ], + [ + "Ġed", + "ge" + ], + [ + "Ġdepend", + "able" + ], + [ + "u", + "nd" + ], + [ + "ut", + "h" + ], + [ + "ĠI", + "f" + ], + [ + "Ġcl", + "ay" + ], + [ + "Ġstrong", + "er" + ], + [ + "Ġboss", + "y" + ], + [ + "Ġc", + "udd" + ], + [ + "as", + "er" + ], + [ + "Ġgra", + "ndpa" + ], + [ + "Ġknow", + "s" + ], + [ + "Ġpick", + "ing" + ], + [ + "Ġjo", + "ke" + ], + [ + "Ġboat", + "s" + ], + [ + "Ġba", + "ld" + ], + [ + "Ġmed", + "ic" + ], + [ + "Ġscr", + "at" + ], + [ + "Ġc", + "one" + ], + [ + "Ġp", + "enny" + ], + [ + "Ġle", + "an" + ], + [ + "Ġra", + "ft" + ], + [ + "ugg", + "led" + ], + [ + "Ġsuc", + "ceed" + ], + [ + "Ġob", + "edient" + ], + [ + "Ġhum", + "ble" + ], + [ + "Ġlibr", + "ary" + ], + [ + "R", + "e" + ], + [ + "he", + "ad" + ], + [ + "Ġp", + "h" + ], + [ + "Ġn", + "umb" + ], + [ + "Ġsa", + "uce" + ], + [ + "ĠB", + "ear" + ], + [ + "Ġtr", + "ay" + ], + [ + "Ġproud", + "ly" + ], + [ + "He", + "re" + ], + [ + "Ġsound", + "ed" + ], + [ + "Ġsqu", + "are" + ], + [ + "Ġterri", + "ble" + ], + [ + "Ġcryst", + "al" + ], + [ + "Ġfier", + "ce" + ], + [ + "J", + "ill" + ], + [ + "Ġw", + "itch" + ], + [ + "in", + "ary" + ], + [ + "Ġhapp", + "iness" + ], + [ + "ĠL", + "ola" + ], + [ + "Ġhand", + "ed" + ], + [ + "Ġfr", + "idge" + ], + [ + "f", + "a" + ], + [ + "Ġr", + "ough" + ], + [ + "ent", + "ion" + ], + [ + "Ġch", + "anged" + ], + [ + "Ġjo", + "ined" + ], + [ + "Ġra", + "ven" + ], + [ + "Ġmight", + "y" + ], + [ + "Ġcho", + "se" + ], + [ + "Ġwhe", + "el" + ], + [ + "Ġumbre", + "lla" + ], + [ + "Ġt", + "ag" + ], + [ + "Ġc", + "ri" + ], + [ + "Ġo", + "w" + ], + [ + "ar", + "ry" + ], + [ + "im", + "p" + ], + [ + "ĠB", + "ill" + ], + [ + "Ġhelp", + "s" + ], + [ + "Ġfind", + "ing" + ], + [ + "Ġclean", + "ing" + ], + [ + "Ġsqu", + "ee" + ], + [ + "Ġfire", + "work" + ], + [ + "Ġmedic", + "ine" + ], + [ + "a", + "zz" + ], + [ + "m", + "et" + ], + [ + "Ġplay", + "ful" + ], + [ + "Ġlo", + "cked" + ], + [ + "Ġgo", + "at" + ], + [ + "Ġac", + "cept" + ], + [ + "Ġcomp", + "ass" + ], + [ + "Ġsur", + "f" + ], + [ + "Ġcompet", + "it" + ], + [ + "e", + "xt" + ], + [ + "k", + "n" + ], + [ + "Ġs", + "y" + ], + [ + "Ġs", + "il" + ], + [ + "Ġbe", + "et" + ], + [ + "ĠM", + "ar" + ], + [ + "ur", + "ing" + ], + [ + "Ġch", + "ir" + ], + [ + "ĠE", + "llie" + ], + [ + "Ġball", + "s" + ], + [ + "Åĵ", + "Let" + ], + [ + "Ġbutterf", + "lies" + ], + [ + "Ġgu", + "ard" + ], + [ + "Ġcray", + "on" + ], + [ + "Ġdisco", + "ver" + ], + [ + "Ġpar", + "ade" + ], + [ + "Ġmix", + "ed" + ], + [ + "Ġval", + "u" + ], + [ + "Ġhel", + "met" + ], + [ + "ĠBrown", + "ie" + ], + [ + "E", + "m" + ], + [ + "j", + "ect" + ], + [ + "s", + "qu" + ], + [ + "u", + "d" + ], + [ + "Ġg", + "um" + ], + [ + "at", + "ure" + ], + [ + "Ġis", + "land" + ], + [ + "Ġco", + "st" + ], + [ + "Ġmo", + "le" + ], + [ + "Ġnice", + "ly" + ], + [ + "Ġkind", + "s" + ], + [ + "Ġra", + "ced" + ], + [ + "Ġsto", + "ve" + ], + [ + "Ġlaugh", + "s" + ], + [ + "Ġmar", + "ch" + ], + [ + "Ġfall", + "s" + ], + [ + "Ġpers", + "ist" + ], + [ + "ĠBa", + "by" + ], + [ + "oph", + "ie" + ], + [ + "A", + "lice" + ], + [ + "i", + "ence" + ], + [ + "it", + "o" + ], + [ + "Ġc", + "age" + ], + [ + "Ġg", + "oose" + ], + [ + "Ġco", + "ins" + ], + [ + "Ġtr", + "ump" + ], + [ + "Ġmo", + "squ" + ], + [ + "Ġser", + "ious" + ], + [ + "Ġmosqu", + "ito" + ], + [ + "Ġf", + "lex" + ], + [ + "Ġc", + "ro" + ], + [ + "el", + "on" + ], + [ + "Ġexc", + "la" + ], + [ + "Ġch", + "oose" + ], + [ + "Ġexpl", + "ored" + ], + [ + "Ġun", + "l" + ], + [ + "Ġbrother", + "s" + ], + [ + "Ġgu", + "il" + ], + [ + "Ġnumb", + "ers" + ], + [ + "Ġtrump", + "et" + ], + [ + "Ġ", + "icy" + ], + [ + "Ġmom", + "s" + ], + [ + "ie", + "w" + ], + [ + "Ġv", + "iew" + ], + [ + "Ġen", + "orm" + ], + [ + "Ġfire", + "man" + ], + [ + "ĠMr", + "s" + ], + [ + "Ġpract", + "ice" + ], + [ + "Ġenorm", + "ous" + ], + [ + "w", + "ards" + ], + [ + "Ġs", + "up" + ], + [ + "Ġvo", + "lc" + ], + [ + "Ġmean", + "s" + ], + [ + "Ġpoin", + "ting" + ], + [ + "Ġvalu", + "able" + ], + [ + "Ġflex", + "ible" + ], + [ + "Ġexcla", + "imed" + ], + [ + "Ġguil", + "ty" + ], + [ + "y", + "al" + ], + [ + "Ġc", + "am" + ], + [ + "Ġp", + "ipe" + ], + [ + "is", + "ion" + ], + [ + "ĠM", + "y" + ], + [ + "Ġuse", + "ful" + ], + [ + "Ġfav", + "our" + ], + [ + "Ġcra", + "zy" + ], + [ + "Ġph", + "ot" + ], + [ + "kn", + "own" + ], + [ + "Ġvolc", + "ano" + ], + [ + "Ġwa", + "ke" + ], + [ + "Ġc", + "ity" + ], + [ + "om", + "ed" + ], + [ + "ll", + "ip" + ], + [ + "Ġfor", + "th" + ], + [ + "Ġlo", + "llip" + ], + [ + "Ġse", + "al" + ], + [ + "Ġall", + "owed" + ], + [ + "Ġsc", + "are" + ], + [ + "Ġbright", + "ly" + ], + [ + "Ġmar", + "ble" + ], + [ + "Ġpop", + "ular" + ], + [ + "Ġflash", + "light" + ], + [ + "Ġcompass", + "ion" + ], + [ + "Ġlollip", + "op" + ], + [ + "Ġ", + "U" + ], + [ + "Ġ", + "ing" + ], + [ + "Ġh", + "unt" + ], + [ + "Ġd", + "ull" + ], + [ + "er", + "ry" + ], + [ + "Ġst", + "at" + ], + [ + "Ġshe", + "l" + ], + [ + "!\"", + "." + ], + [ + "Ġun", + "known" + ], + [ + "Ġhat", + "s" + ], + [ + "Ġsho", + "ot" + ], + [ + "red", + "ient" + ], + [ + "Ġbelong", + "ed" + ], + [ + "Ġfavour", + "ite" + ], + [ + "Ġing", + "redient" + ], + [ + "Ġp", + "il" + ], + [ + "Ġlo", + "yal" + ], + [ + "Ġbra", + "ce" + ], + [ + "ĠWhen", + "ever" + ], + [ + "Ġdream", + "s" + ], + [ + "Ġkick", + "ed" + ], + [ + "Ġseed", + "s" + ], + [ + "Ġtwirl", + "ed" + ], + [ + "Ġsup", + "er" + ], + [ + "t", + "ime" + ], + [ + "Ġf", + "ine" + ], + [ + "Ġbe", + "es" + ], + [ + "Ġso", + "fa" + ], + [ + "Ġsh", + "ark" + ], + [ + "ĠM", + "ike" + ], + [ + "Ġan", + "x" + ], + [ + "Ġsad", + "ly" + ], + [ + "Ġshow", + "ing" + ], + [ + "Ġway", + "s" + ], + [ + "Ġtruck", + "s" + ], + [ + "Ġdelic", + "ate" + ], + [ + "ation", + "s" + ], + [ + "Ġrep", + "e" + ], + [ + "Ġimpress", + "ive" + ], + [ + "Ġpour", + "ed" + ], + [ + "Ġpir", + "ate" + ], + [ + "Ġwhisp", + "ered" + ], + [ + "Ġharm", + "less" + ], + [ + "a", + "ff" + ], + [ + "Ġt", + "urt" + ], + [ + "Ġt", + "ied" + ], + [ + "ar", + "oo" + ], + [ + "ad", + "o" + ], + [ + "art", + "h" + ], + [ + "Ġgra", + "ce" + ], + [ + "Ġhand", + "le" + ], + [ + "Ġen", + "c" + ], + [ + "Ġdel", + "iver" + ], + [ + "Ben", + "ny" + ], + [ + "Ġru", + "shed" + ], + [ + "Ġus", + "ing" + ], + [ + "ang", + "aroo" + ], + [ + "Ġshap", + "e" + ], + [ + "Ġbus", + "hes" + ], + [ + "Ġpersist", + "ent" + ], + [ + "Ġg", + "or" + ], + [ + "ĠS", + "p" + ], + [ + "Ġk", + "angaroo" + ], + [ + "Ġan", + "g" + ], + [ + "Ġoff", + "ice" + ], + [ + "Ġanx", + "ious" + ], + [ + "G", + "o" + ], + [ + "al", + "ous" + ], + [ + "Ġsp", + "ell" + ], + [ + "ĠW", + "hile" + ], + [ + "ia", + "ble" + ], + [ + "Ġpot", + "ato" + ], + [ + "Ġje", + "alous" + ], + [ + "Ġwra", + "pped" + ], + [ + "Ġf", + "our" + ], + [ + "Ġc", + "urt" + ], + [ + "Ġlo", + "g" + ], + [ + "Ġco", + "al" + ], + [ + "Ġrel", + "iable" + ], + [ + "Ġcurt", + "ain" + ], + [ + "D", + "aisy" + ], + [ + "Ġs", + "udden" + ], + [ + "mb", + "ol" + ], + [ + "Åĵ", + "Yes" + ], + [ + "ac", + "hes" + ], + [ + "ĠP", + "e" + ], + [ + "Ġpain", + "ted" + ], + [ + "ĠTo", + "by" + ], + [ + "Ġce", + "re" + ], + [ + "g", + "u" + ], + [ + "Ġt", + "ummy" + ], + [ + "Ġth", + "ose" + ], + [ + "ous", + "es" + ], + [ + "udd", + "y" + ], + [ + "Ġcu", + "sh" + ], + [ + "Ġmet", + "al" + ], + [ + "Ġsy", + "mbol" + ], + [ + "Ġbeet", + "le" + ], + [ + "Ġcam", + "era" + ], + [ + "Ġh", + "ouses" + ], + [ + "il", + "t" + ], + [ + "Ġcelebr", + "ate" + ], + [ + "Ġdelight", + "ed" + ], + [ + "Ġgor", + "illa" + ], + [ + "Ġto", + "oth" + ], + [ + "Ġth", + "irst" + ], + [ + "ke", + "ep" + ], + [ + "Ġsh", + "iver" + ], + [ + "Ġre", + "ward" + ], + [ + "Ġsp", + "ace" + ], + [ + "Ġtra", + "vel" + ], + [ + "Ġru", + "bbed" + ], + [ + "Ġprin", + "t" + ], + [ + "Ġdri", + "ving" + ], + [ + "Ġmar", + "ry" + ], + [ + "Ġwar", + "ned" + ], + [ + "Ġsold", + "ier" + ], + [ + "Ġmotor", + "cy" + ], + [ + "c", + "ut" + ], + [ + "Ġs", + "on" + ], + [ + "Ġto", + "r" + ], + [ + "ar", + "lie" + ], + [ + "Ġher", + "o" + ], + [ + "ie", + "f" + ], + [ + "Ġcar", + "p" + ], + [ + "Ġexcited", + "ly" + ], + [ + "Ġte", + "mp" + ], + [ + "ĠP", + "ete" + ], + [ + "ĠBob", + "o" + ], + [ + "co", + "d" + ], + [ + "Ġhun", + "g" + ], + [ + "Ġgrow", + "ing" + ], + [ + "up", + "id" + ], + [ + "Ġthin", + "king" + ], + [ + "Ġtun", + "n" + ], + [ + "W", + "hile" + ], + [ + "c", + "u" + ], + [ + "i", + "pp" + ], + [ + "Ġh", + "ay" + ], + [ + "Ġy", + "et" + ], + [ + "Ġv", + "ine" + ], + [ + "Ġac", + "orn" + ], + [ + "icy", + "cle" + ], + [ + "Ġpre", + "p" + ], + [ + "Ġremind", + "ed" + ], + [ + "Ġgas", + "ped" + ], + [ + "Ġflut", + "e" + ], + [ + "Ġord", + "inary" + ], + [ + "Ġcro", + "cod" + ], + [ + "Ġa", + "mb" + ], + [ + "Ġc", + "ur" + ], + [ + "ĠB", + "et" + ], + [ + "ĠM", + "andy" + ], + [ + "Ġpu", + "p" + ], + [ + "Ġcreature", + "s" + ], + [ + "Ġhur", + "ry" + ], + [ + "Ġspoil", + "ed" + ], + [ + "Ġt", + "imes" + ], + [ + "Ġm", + "is" + ], + [ + "Ġr", + "ice" + ], + [ + "Ġst", + "upid" + ], + [ + "Ġsh", + "ine" + ], + [ + "Ġre", + "sist" + ], + [ + "ug", + "ged" + ], + [ + "Ġla", + "w" + ], + [ + "Ġsk", + "ull" + ], + [ + "co", + "a" + ], + [ + "Ġmiss", + "ing" + ], + [ + "Ġwhe", + "at" + ], + [ + "Ġsupp", + "ort" + ], + [ + "Ġingredient", + "s" + ], + [ + "a", + "id" + ], + [ + "l", + "oo" + ], + [ + "Ġc", + "ase" + ], + [ + "er", + "y" + ], + [ + "Ġwas", + "hed" + ], + [ + "Ġdo", + "ve" + ], + [ + "Ġsp", + "illed" + ], + [ + "Ġsc", + "oot" + ], + [ + "Ġme", + "al" + ], + [ + "Ġmu", + "le" + ], + [ + "ĠD", + "ucky" + ], + [ + "Ġmo", + "der" + ], + [ + "ars", + "h" + ], + [ + "Åĵ", + "What" + ], + [ + "Ġtri", + "ck" + ], + [ + "Ġset", + "t" + ], + [ + "ĠTweet", + "ie" + ], + [ + "Ġboun", + "ce" + ], + [ + "Ġfil", + "thy" + ], + [ + "Ġb", + "ase" + ], + [ + "Ġh", + "oop" + ], + [ + "Ġf", + "re" + ], + [ + "Ġst", + "umbled" + ], + [ + "Ġso", + "ar" + ], + [ + "Ġwe", + "al" + ], + [ + "Ġsee", + "m" + ], + [ + "Ġwor", + "d" + ], + [ + "Ġla", + "b" + ], + [ + "ĠD", + "ave" + ], + [ + "ĠF", + "r" + ], + [ + "Ġbad", + "ly" + ], + [ + "Ġhair", + "y" + ], + [ + "Ġcra", + "wl" + ], + [ + "Ġlive", + "ly" + ], + [ + "Ġstep", + "s" + ], + [ + "ÅĵLet", + "â" + ], + [ + "Ġbrace", + "let" + ], + [ + "Ġcere", + "al" + ], + [ + "Ġthirst", + "y" + ], + [ + "Ġcarp", + "et" + ], + [ + "Ġw", + "ool" + ], + [ + "it", + "her" + ], + [ + "Ġgo", + "al" + ], + [ + "Ġback", + "pack" + ], + [ + "Ġtr", + "uth" + ], + [ + "Ġcom", + "pl" + ], + [ + "Ġche", + "ap" + ], + [ + "Ġdis", + "g" + ], + [ + "Ġplac", + "ed" + ], + [ + "Ġear", + "ly" + ], + [ + "Ġdecor", + "ate" + ], + [ + "Ġnut", + "s" + ], + [ + "O", + "w" + ], + [ + "c", + "ream" + ], + [ + "Ġb", + "ump" + ], + [ + "Ġthem", + "selves" + ], + [ + "ic", + "es" + ], + [ + "Ġhelp", + "less" + ], + [ + "Ġcl", + "um" + ], + [ + "Ġsc", + "atter" + ], + [ + "Ġsc", + "ale" + ], + [ + "Ġnew", + "s" + ], + [ + "ust", + "ing" + ], + [ + "Ġen", + "v" + ], + [ + "âĤ¬â", + "Ģ" + ], + [ + "Ġtell", + "ing" + ], + [ + "Ġswing", + "ing" + ], + [ + "Ġspark", + "led" + ], + [ + "Ġcarry", + "ing" + ], + [ + "gg", + "ing" + ], + [ + "in", + "es" + ], + [ + "ri", + "c" + ], + [ + "Ġfriends", + "hip" + ], + [ + "Ġcare", + "less" + ], + [ + "Ġsn", + "e" + ], + [ + "Ġdoes", + "n" + ], + [ + "Ġfall", + "ing" + ], + [ + "Ġcomp", + "ut" + ], + [ + "Ġpa", + "le" + ], + [ + "Ġpee", + "ked" + ], + [ + "Ġcompassion", + "ate" + ], + [ + "c", + "om" + ], + [ + "v", + "ice" + ], + [ + "Ġs", + "end" + ], + [ + "Ġst", + "age" + ], + [ + "Ġme", + "m" + ], + [ + "Ġch", + "alk" + ], + [ + "Ġch", + "asing" + ], + [ + "Ġpe", + "bble" + ], + [ + "ĠEvery", + "thing" + ], + [ + "Ġcu", + "be" + ], + [ + "Ġpig", + "e" + ], + [ + "Ġcou", + "rage" + ], + [ + "Ġfing", + "ers" + ], + [ + "Ġturt", + "le" + ], + [ + "b", + "led" + ], + [ + "Ġs", + "ink" + ], + [ + "Ġb", + "icycle" + ], + [ + "Ġsh", + "aking" + ], + [ + "Ġsleep", + "ing" + ], + [ + "Ġneighb", + "our" + ], + [ + "Ġmoder", + "n" + ], + [ + "Ġdisg", + "usting" + ], + [ + "Ġclum", + "sy" + ], + [ + "g", + "est" + ], + [ + "Ġh", + "ook" + ], + [ + "Ġc", + "her" + ], + [ + "Ġn", + "ail" + ], + [ + "Ġbig", + "gest" + ], + [ + "ic", + "hes" + ], + [ + "Ġbu", + "l" + ], + [ + "Ġend", + "ing" + ], + [ + "Ġde", + "ad" + ], + [ + "Ġtri", + "ang" + ], + [ + "ul", + "ance" + ], + [ + "Ġspo", + "ke" + ], + [ + "Ġpract", + "iced" + ], + [ + "Ġamb", + "ulance" + ], + [ + "Ġcomput", + "er" + ], + [ + "i", + "bb" + ], + [ + "Ġl", + "amp" + ], + [ + "Ġst", + "ation" + ], + [ + "Ġch", + "ar" + ], + [ + "Ġwall", + "et" + ], + [ + "ĠTo", + "day" + ], + [ + "Ġpack", + "age" + ], + [ + "Ġmic", + "rop" + ], + [ + "Ġcush", + "ion" + ], + [ + "keep", + "er" + ], + [ + "Ġcrocod", + "ile" + ], + [ + "Ġmicrop", + "hone" + ], + [ + "u", + "nder" + ], + [ + "ĠA", + "lex" + ], + [ + "Ġthought", + "ful" + ], + [ + "Ġjo", + "lly" + ], + [ + "Ġpen", + "gu" + ], + [ + "Ġpump", + "kin" + ], + [ + "Ġcost", + "um" + ], + [ + "Ġf", + "ive" + ], + [ + "Ġd", + "ough" + ], + [ + "Ġp", + "ony" + ], + [ + "Ġo", + "l" + ], + [ + "im", + "i" + ], + [ + "Ġst", + "airs" + ], + [ + "Ġkn", + "ock" + ], + [ + "Ġgra", + "nd" + ], + [ + "Ġint", + "ell" + ], + [ + "Ġun", + "iver" + ], + [ + "Ġbright", + "er" + ], + [ + "Ġopen", + "s" + ], + [ + "Ġdream", + "ing" + ], + [ + "Ġboun", + "ced" + ], + [ + "Ġquest", + "ion" + ], + [ + "Ġstat", + "ue" + ], + [ + "Ġw", + "ine" + ], + [ + "is", + "k" + ], + [ + "ĠA", + "lice" + ], + [ + "Ġco", + "coa" + ], + [ + "ĠYou", + "r" + ], + [ + "Ġpe", + "pper" + ], + [ + "Ġbeaut", + "y" + ], + [ + "Ġper", + "m" + ], + [ + "Ġpain", + "ting" + ], + [ + "Ġsho", + "e" + ], + [ + "Ġele", + "v" + ], + [ + "Every", + "one" + ], + [ + "hin", + "o" + ], + [ + "Ġweal", + "thy" + ], + [ + "Ġintell", + "ig" + ], + [ + "n", + "oon" + ], + [ + "Ġs", + "uit" + ], + [ + "Ġm", + "ild" + ], + [ + "Ġn", + "ature" + ], + [ + "an", + "s" + ], + [ + "Ġhapp", + "ier" + ], + [ + "Ġne", + "at" + ], + [ + "Ġstart", + "s" + ], + [ + "uck", + "ed" + ], + [ + "Ġafter", + "noon" + ], + [ + "Ġtor", + "n" + ], + [ + "Ġelev", + "ator" + ], + [ + "g", + "en" + ], + [ + "t", + "i" + ], + [ + "Ġt", + "en" + ], + [ + "Ġh", + "arsh" + ], + [ + "Ġp", + "ed" + ], + [ + "at", + "ient" + ], + [ + "or", + "ant" + ], + [ + "Ġre", + "ce" + ], + [ + "Ġj", + "et" + ], + [ + "ill", + "ie" + ], + [ + "ĠR", + "ed" + ], + [ + "Ġread", + "ing" + ], + [ + "Ġsear", + "ching" + ], + [ + "Ġbasket", + "ball" + ], + [ + "Ġbar", + "ber" + ], + [ + "Ġspe", + "ed" + ], + [ + "S", + "ee" + ], + [ + "a", + "wn" + ], + [ + "Ġst", + "ack" + ], + [ + "ĠM", + "ummy" + ], + [ + "Ġclo", + "ck" + ], + [ + "ĠG", + "ive" + ], + [ + "Ġmag", + "n" + ], + [ + "Ġlea", + "ving" + ], + [ + "Ġcup", + "board" + ], + [ + "Ġdes", + "ign" + ], + [ + "Ġslid", + "es" + ], + [ + "Ġsandw", + "iches" + ], + [ + "Ġcand", + "le" + ], + [ + "aul", + "if" + ], + [ + "Ġrid", + "ing" + ], + [ + "Ġfre", + "sh" + ], + [ + "o", + "ppy" + ], + [ + "Ġsh", + "ocked" + ], + [ + "Ġen", + "er" + ], + [ + "Ġjump", + "s" + ], + [ + "Ġhair", + "cut" + ], + [ + "Ġrespect", + "ful" + ], + [ + "Ġcoo", + "king" + ], + [ + "Ġtick", + "et" + ], + [ + "Ġenc", + "ou" + ], + [ + "Ġtunn", + "el" + ], + [ + "aulif", + "lower" + ], + [ + "G", + "ive" + ], + [ + "Ġs", + "le" + ], + [ + "re", + "en" + ], + [ + "Ġm", + "ay" + ], + [ + "Ġth", + "ief" + ], + [ + "Ġy", + "awn" + ], + [ + "Ġfor", + "t" + ], + [ + "Ġal", + "ert" + ], + [ + "Ġsor", + "ts" + ], + [ + "Ġimp", + "atient" + ], + [ + "Ġcre", + "ate" + ], + [ + "ail", + "able" + ], + [ + "Ġwhe", + "els" + ], + [ + "Ġval", + "ue" + ], + [ + "Ġpri", + "ze" + ], + [ + "M", + "ary" + ], + [ + "y", + "e" + ], + [ + "he", + "t" + ], + [ + "Ġs", + "ent" + ], + [ + "Ġb", + "ent" + ], + [ + "in", + "ce" + ], + [ + "Ġc", + "omet" + ], + [ + "Ġc", + "amp" + ], + [ + "Ġc", + "auliflower" + ], + [ + "Ġm", + "elon" + ], + [ + "Ġg", + "em" + ], + [ + "Ġit", + "self" + ], + [ + "ir", + "on" + ], + [ + "Ġmu", + "sh" + ], + [ + "Ġmonkey", + "s" + ], + [ + "Ġsal", + "ad" + ], + [ + "Ġdest", + "ro" + ], + [ + "Ġresc", + "ue" + ], + [ + "Ġscoot", + "er" + ], + [ + "Ġintellig", + "ent" + ], + [ + "a", + "z" + ], + [ + "Ġd", + "ug" + ], + [ + "il", + "ing" + ], + [ + "an", + "ger" + ], + [ + "Ġr", + "oof" + ], + [ + "Ġr", + "hino" + ], + [ + "Ġsc", + "old" + ], + [ + "ag", + "het" + ], + [ + "Ġexp", + "er" + ], + [ + "Ġbark", + "ing" + ], + [ + "Ġbanan", + "as" + ], + [ + "Ġign", + "orant" + ], + [ + "Ġmic", + "ro" + ], + [ + "Ġav", + "ailable" + ], + [ + "Ġgrace", + "ful" + ], + [ + "Ġmotorcy", + "cle" + ], + [ + "aghet", + "ti" + ], + [ + "h", + "ood" + ], + [ + "ĠS", + "n" + ], + [ + "Ġe", + "arth" + ], + [ + "Ġbo", + "ots" + ], + [ + "Ġsp", + "aghetti" + ], + [ + "um", + "mer" + ], + [ + "Ġag", + "ree" + ], + [ + "Ġbu", + "ll" + ], + [ + "Ġany", + "where" + ], + [ + "ĠG", + "eorge" + ], + [ + "Ġbeh", + "ave" + ], + [ + "ĠK", + "im" + ], + [ + "Ġpa", + "id" + ], + [ + "sc", + "ope" + ], + [ + "Ġstret", + "ched" + ], + [ + "Ġfear", + "ful" + ], + [ + "Ġav", + "o" + ], + [ + "Ġmicro", + "scope" + ], + [ + "I", + "f" + ], + [ + "i", + "o" + ], + [ + "Ġnot", + "eb" + ], + [ + "ĠE", + "ventually" + ], + [ + "Ġold", + "er" + ], + [ + "Ġsn", + "iff" + ], + [ + "Ġad", + "vice" + ], + [ + "Ġstop", + "s" + ], + [ + "Ġperf", + "orm" + ], + [ + "Ġfur", + "ther" + ], + [ + "Ġenv", + "ious" + ], + [ + "Ġpige", + "on" + ], + [ + "Ġmush", + "room" + ], + [ + "Ġnoteb", + "ook" + ], + [ + "u", + "bby" + ], + [ + "Ġp", + "un" + ], + [ + "at", + "s" + ], + [ + "Ġn", + "urse" + ], + [ + "or", + "able" + ], + [ + "Ġth", + "under" + ], + [ + "Ġcl", + "apping" + ], + [ + "Ġv", + "ide" + ], + [ + "Ġbu", + "ilt" + ], + [ + "ĠC", + "l" + ], + [ + "Ġun", + "com" + ], + [ + "Ġshout", + "ing" + ], + [ + "Ġar", + "row" + ], + [ + "Ġdraw", + "er" + ], + [ + "Ġpro", + "v" + ], + [ + "fort", + "able" + ], + [ + "Ġtom", + "ato" + ], + [ + "Ġmeet", + "ing" + ], + [ + "Ġob", + "ject" + ], + [ + "Ġyog", + "urt" + ], + [ + "Ġtemp", + "le" + ], + [ + "Ġuncom", + "fortable" + ], + [ + "e", + "f" + ], + [ + "u", + "es" + ], + [ + "ĠT", + "ony" + ], + [ + "Ġloo", + "p" + ], + [ + "Ġch", + "ubby" + ], + [ + "Ġsw", + "itch" + ], + [ + "Ġsk", + "ip" + ], + [ + "Ġad", + "orable" + ], + [ + "Åĵ", + "It" + ], + [ + "Ġcount", + "ing" + ], + [ + "Ġcoo", + "ked" + ], + [ + "Ġp", + "ist" + ], + [ + "ar", + "ian" + ], + [ + "en", + "ry" + ], + [ + "Ġst", + "ared" + ], + [ + "Ġsh", + "ut" + ], + [ + "ic", + "op" + ], + [ + "Ġqu", + "ite" + ], + [ + "Ġsn", + "uggled" + ], + [ + "ĠGra", + "ndpa" + ], + [ + "Ġtrou", + "bled" + ], + [ + "Ġgigg", + "le" + ], + [ + "Ġblow", + "ing" + ], + [ + "Ġhel", + "icop" + ], + [ + "Ġfis", + "her" + ], + [ + "Ġpist", + "ol" + ], + [ + "Ġhelicop", + "ter" + ], + [ + "p", + "ar" + ], + [ + "ĠS", + "ophie" + ], + [ + "oo", + "ped" + ], + [ + "ip", + "s" + ], + [ + "Ġnow", + "here" + ], + [ + "Ġforg", + "ave" + ], + [ + "Ġmar", + "ched" + ], + [ + "Ġtreasure", + "s" + ], + [ + "ho", + "st" + ], + [ + "Ġpass", + "port" + ], + [ + "Ġstir", + "red" + ], + [ + "Ġpus", + "hing" + ], + [ + "Ġcompetit", + "ive" + ], + [ + "Ġdestro", + "y" + ], + [ + "Y", + "ay" + ], + [ + "a", + "ked" + ], + [ + "o", + "e" + ], + [ + "t", + "ter" + ], + [ + "Ġg", + "ir" + ], + [ + "Ġg", + "host" + ], + [ + "an", + "ic" + ], + [ + "ke", + "let" + ], + [ + "ht", + "ub" + ], + [ + "Ġre", + "ef" + ], + [ + "Ġse", + "par" + ], + [ + "op", + "ard" + ], + [ + "pl", + "ay" + ], + [ + "Ġen", + "vel" + ], + [ + "Ġbre", + "at" + ], + [ + "Ġrep", + "air" + ], + [ + "Ġfoot", + "ball" + ], + [ + "Ġbat", + "htub" + ], + [ + "Ġrubb", + "er" + ], + [ + "Ġang", + "el" + ], + [ + "Ġtriang", + "le" + ], + [ + "Ġl", + "ed" + ], + [ + "Ġm", + "ill" + ], + [ + "ĠS", + "h" + ], + [ + "ch", + "anic" + ], + [ + "Ġnam", + "es" + ], + [ + "Ġle", + "opard" + ], + [ + "Ġno", + "body" + ], + [ + "Ġany", + "way" + ], + [ + "uc", + "t" + ], + [ + "ct", + "us" + ], + [ + "Ġshould", + "n" + ], + [ + "Ġz", + "eb" + ], + [ + "Ġgi", + "ven" + ], + [ + "Ġhold", + "s" + ], + [ + "Ġbar", + "rel" + ], + [ + "Ġband", + "age" + ], + [ + "Ġent", + "h" + ], + [ + "Dad", + "dy" + ], + [ + "Ġzeb", + "ra" + ], + [ + "i", + "ves" + ], + [ + "u", + "be" + ], + [ + "y", + "ear" + ], + [ + "Ġp", + "or" + ], + [ + "Ġp", + "and" + ], + [ + "Ġo", + "tter" + ], + [ + "Ġr", + "ibb" + ], + [ + "Ġsh", + "ield" + ], + [ + "Ġme", + "chanic" + ], + [ + "Ġdon", + "â" + ], + [ + "Ġdis", + "ag" + ], + [ + "Ġdis", + "play" + ], + [ + "Ġmusic", + "ian" + ], + [ + "cer", + "os" + ], + [ + "Ġspe", + "ak" + ], + [ + "Ġca", + "b" + ], + [ + "Ġca", + "ctus" + ], + [ + "ĠBet", + "sy" + ], + [ + "Ġbul", + "b" + ], + [ + "C", + "h" + ], + [ + "e", + "ath" + ], + [ + "i", + "kes" + ], + [ + "x", + "y" + ], + [ + "Ġt", + "ube" + ], + [ + "Ġa", + "head" + ], + [ + "Ġs", + "ummer" + ], + [ + "Ġs", + "kelet" + ], + [ + "Ġp", + "urse" + ], + [ + "Ġkn", + "ob" + ], + [ + "Ġdidn", + "â" + ], + [ + "Ġcouldn", + "â" + ], + [ + "Ġcol", + "ours" + ], + [ + "Ġwindow", + "s" + ], + [ + "oom", + "y" + ], + [ + "Ġgif", + "ted" + ], + [ + "Ġcart", + "oon" + ], + [ + "Ġrhino", + "ceros" + ], + [ + "g", + "l" + ], + [ + "he", + "art" + ], + [ + "Ġs", + "ize" + ], + [ + "Ġb", + "et" + ], + [ + "ĠM", + "iss" + ], + [ + "Ġsc", + "rew" + ], + [ + "Ġfl", + "our" + ], + [ + "Ġbl", + "in" + ], + [ + "ap", + "a" + ], + [ + "Ġdis", + "hes" + ], + [ + "Ġrain", + "ing" + ], + [ + "Ġhop", + "ing" + ], + [ + "Ġsho", + "vel" + ], + [ + "Ġmin", + "t" + ], + [ + "Ġjelly", + "fish" + ], + [ + "Ġswe", + "ater" + ], + [ + "Ġchar", + "ming" + ], + [ + "Ġpengu", + "in" + ], + [ + "Ġenvel", + "ope" + ], + [ + "Ġenth", + "us" + ], + [ + "Ġcab", + "in" + ], + [ + "re", + "c" + ], + [ + "Ġp", + "ant" + ], + [ + "Ġth", + "read" + ], + [ + "Ġin", + "se" + ], + [ + "Ġst", + "aff" + ], + [ + "Ġli", + "z" + ], + [ + "Ġun", + "pack" + ], + [ + "Ġwr", + "iting" + ], + [ + "Ġtra", + "sh" + ], + [ + "Ġroll", + "ing" + ], + [ + "Ġwaff", + "le" + ], + [ + "Ġ", + "iron" + ], + [ + "Ġw", + "ing" + ], + [ + "er", + "able" + ], + [ + "Ġshe", + "et" + ], + [ + "ĠB", + "uddy" + ], + [ + "Ġsh", + "adow" + ], + [ + "Ġro", + "ared" + ], + [ + "Ġmu", + "ff" + ], + [ + "Ġair", + "port" + ], + [ + "Ġce", + "iling" + ], + [ + "Ġatt", + "ic" + ], + [ + "Ġorgan", + "ize" + ], + [ + "Ġstret", + "ch" + ], + [ + "Ġgrap", + "es" + ], + [ + "Ġo", + "bs" + ], + [ + "ĠA", + "re" + ], + [ + "ĠW", + "ould" + ], + [ + "ĠIn", + "st" + ], + [ + "Ġpi", + "ano" + ], + [ + "Ġsal", + "t" + ], + [ + "Ġmis", + "erable" + ], + [ + "Ġc", + "ord" + ], + [ + "Ġd", + "ess" + ], + [ + "Ġre", + "fused" + ], + [ + "Ġsc", + "reen" + ], + [ + "ĠD", + "an" + ], + [ + "Ġstr", + "ugg" + ], + [ + "Ġneed", + "le" + ], + [ + "Ġbear", + "s" + ], + [ + "ĠAnd", + "y" + ], + [ + "Ġhead", + "ed" + ], + [ + "Ġstick", + "y" + ], + [ + "Ġfree", + "z" + ], + [ + "Ġzoom", + "ed" + ], + [ + "Ġimag", + "ined" + ], + [ + "Ġmed", + "al" + ], + [ + "Ġmagn", + "et" + ], + [ + "rec", + "i" + ], + [ + "Ġobs", + "er" + ], + [ + "Ġ", + "er" + ], + [ + "Ġli", + "ves" + ], + [ + "ĠA", + "l" + ], + [ + "ĠP", + "at" + ], + [ + "Ġapp", + "reci" + ], + [ + "ass", + "es" + ], + [ + "Ġsail", + "or" + ], + [ + "Ġminut", + "e" + ], + [ + "Ġfisher", + "man" + ], + [ + "v", + "ision" + ], + [ + "Ġr", + "at" + ], + [ + "al", + "ed" + ], + [ + "Ġbr", + "us" + ], + [ + "Ġch", + "ance" + ], + [ + "Ġpo", + "st" + ], + [ + "Ġsun", + "gl" + ], + [ + "Ġche", + "ek" + ], + [ + "ree", + "ze" + ], + [ + "Ġde", + "af" + ], + [ + "Ġru", + "les" + ], + [ + "Ġfrog", + "s" + ], + [ + "Ġair", + "pl" + ], + [ + "ĠGo", + "d" + ], + [ + "Ġstrawber", + "ry" + ], + [ + "Ġsle", + "pt" + ], + [ + "Ġliz", + "ard" + ], + [ + "Ġsungl", + "asses" + ], + [ + "Ġl", + "uck" + ], + [ + "Ġin", + "f" + ], + [ + "Ġbe", + "ep" + ], + [ + "ĠB", + "unny" + ], + [ + "Ġme", + "as" + ], + [ + "Ġro", + "d" + ], + [ + "Ġor", + "der" + ], + [ + "Ġrest", + "less" + ], + [ + "Ġsand", + "box" + ], + [ + "Ġmess", + "age" + ], + [ + "ĠJoe", + "y" + ], + [ + "Ġcomp", + "let" + ], + [ + "Ġatt", + "ract" + ], + [ + "Ġgold", + "en" + ], + [ + "Gra", + "ndma" + ], + [ + "Ġpil", + "ot" + ], + [ + "Ġpor", + "ch" + ], + [ + "L", + "ittle" + ], + [ + "l", + "er" + ], + [ + "s", + "es" + ] + ] + } +} \ No newline at end of file diff --git a/outio/mlp-linear-9L_run/checkpoint-750/tokenizer_config.json b/outio/mlp-linear-9L_run/checkpoint-750/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..f1578d3b6947e13e4001d559b79d4944be52b9c1 --- /dev/null +++ b/outio/mlp-linear-9L_run/checkpoint-750/tokenizer_config.json @@ -0,0 +1,13 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": "<|endoftext|>", + "eos_token": "<|endoftext|>", + "errors": "replace", + "is_local": false, + "local_files_only": false, + "model_max_length": 1024, + "pad_token": "<|endoftext|>", + "tokenizer_class": "GPT2Tokenizer", + "unk_token": "<|endoftext|>" +} diff --git a/outio/mlp-linear-9L_run/checkpoint-750/trainer_state.json b/outio/mlp-linear-9L_run/checkpoint-750/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..90a9237e7fa52c67b5f8041fc1f7b3b5bf5d48fd --- /dev/null +++ b/outio/mlp-linear-9L_run/checkpoint-750/trainer_state.json @@ -0,0 +1,413 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0107853050219076, + "eval_steps": 50, + "global_step": 750, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.026963262554769128, + "grad_norm": 19.875, + "learning_rate": 0.0005, + "loss": 120.13565673828126, + "step": 20 + }, + { + "epoch": 0.053926525109538256, + "grad_norm": 24.625, + "learning_rate": 0.0005, + "loss": 102.05034790039062, + "step": 40 + }, + { + "epoch": 0.06740815638692282, + "eval_loss": 5.665860652923584, + "eval_runtime": 9.5524, + "eval_samples_per_second": 997.346, + "eval_steps_per_second": 12.562, + "step": 50 + }, + { + "epoch": 0.08088978766430738, + "grad_norm": 18.125, + "learning_rate": 0.0005, + "loss": 90.97202758789062, + "step": 60 + }, + { + "epoch": 0.10785305021907651, + "grad_norm": 40.75, + "learning_rate": 0.0005, + "loss": 83.10302734375, + "step": 80 + }, + { + "epoch": 0.13481631277384565, + "grad_norm": 29.625, + "learning_rate": 0.0005, + "loss": 77.51190185546875, + "step": 100 + }, + { + "epoch": 0.13481631277384565, + "eval_loss": 4.689935684204102, + "eval_runtime": 9.1111, + "eval_samples_per_second": 1045.652, + "eval_steps_per_second": 13.171, + "step": 100 + }, + { + "epoch": 0.16177957532861476, + "grad_norm": 19.5, + "learning_rate": 0.0005, + "loss": 73.29612426757812, + "step": 120 + }, + { + "epoch": 0.1887428378833839, + "grad_norm": 25.25, + "learning_rate": 0.0005, + "loss": 70.37640380859375, + "step": 140 + }, + { + "epoch": 0.20222446916076844, + "eval_loss": 4.263797760009766, + "eval_runtime": 9.4931, + "eval_samples_per_second": 1003.568, + "eval_steps_per_second": 12.641, + "step": 150 + }, + { + "epoch": 0.21570610043815303, + "grad_norm": 14.625, + "learning_rate": 0.0005, + "loss": 68.2427978515625, + "step": 160 + }, + { + "epoch": 0.24266936299292213, + "grad_norm": 13.0625, + "learning_rate": 0.0005, + "loss": 66.13766479492188, + "step": 180 + }, + { + "epoch": 0.2696326255476913, + "grad_norm": 18.5, + "learning_rate": 0.0005, + "loss": 63.81178588867188, + "step": 200 + }, + { + "epoch": 0.2696326255476913, + "eval_loss": 3.9343974590301514, + "eval_runtime": 9.3868, + "eval_samples_per_second": 1014.939, + "eval_steps_per_second": 12.784, + "step": 200 + }, + { + "epoch": 0.2965958881024604, + "grad_norm": 16.125, + "learning_rate": 0.0005, + "loss": 62.16768188476563, + "step": 220 + }, + { + "epoch": 0.3235591506572295, + "grad_norm": 21.25, + "learning_rate": 0.0005, + "loss": 61.1105712890625, + "step": 240 + }, + { + "epoch": 0.33704078193461406, + "eval_loss": 3.748317241668701, + "eval_runtime": 9.4777, + "eval_samples_per_second": 1005.204, + "eval_steps_per_second": 12.661, + "step": 250 + }, + { + "epoch": 0.3505224132119987, + "grad_norm": 42.25, + "learning_rate": 0.0005, + "loss": 60.05653686523438, + "step": 260 + }, + { + "epoch": 0.3774856757667678, + "grad_norm": 27.5, + "learning_rate": 0.0005, + "loss": 59.25806884765625, + "step": 280 + }, + { + "epoch": 0.4044489383215369, + "grad_norm": 14.375, + "learning_rate": 0.0005, + "loss": 58.334844970703124, + "step": 300 + }, + { + "epoch": 0.4044489383215369, + "eval_loss": 3.6137855052948, + "eval_runtime": 9.6654, + "eval_samples_per_second": 985.685, + "eval_steps_per_second": 12.415, + "step": 300 + }, + { + "epoch": 0.43141220087630605, + "grad_norm": 32.75, + "learning_rate": 0.0005, + "loss": 57.341180419921876, + "step": 320 + }, + { + "epoch": 0.45837546343107516, + "grad_norm": 12.5625, + "learning_rate": 0.0005, + "loss": 56.699591064453124, + "step": 340 + }, + { + "epoch": 0.4718570947084597, + "eval_loss": 3.492727518081665, + "eval_runtime": 9.4437, + "eval_samples_per_second": 1008.824, + "eval_steps_per_second": 12.707, + "step": 350 + }, + { + "epoch": 0.48533872598584427, + "grad_norm": 21.5, + "learning_rate": 0.0005, + "loss": 55.840753173828126, + "step": 360 + }, + { + "epoch": 0.5123019885406134, + "grad_norm": 14.5625, + "learning_rate": 0.0005, + "loss": 55.377203369140624, + "step": 380 + }, + { + "epoch": 0.5392652510953826, + "grad_norm": 14.6875, + "learning_rate": 0.0005, + "loss": 54.936090087890626, + "step": 400 + }, + { + "epoch": 0.5392652510953826, + "eval_loss": 3.41583251953125, + "eval_runtime": 9.1837, + "eval_samples_per_second": 1037.382, + "eval_steps_per_second": 13.067, + "step": 400 + }, + { + "epoch": 0.5662285136501517, + "grad_norm": 22.0, + "learning_rate": 0.0005, + "loss": 54.31259765625, + "step": 420 + }, + { + "epoch": 0.5931917762049208, + "grad_norm": 22.5, + "learning_rate": 0.0005, + "loss": 53.894476318359374, + "step": 440 + }, + { + "epoch": 0.6066734074823054, + "eval_loss": 3.3446545600891113, + "eval_runtime": 9.6871, + "eval_samples_per_second": 983.477, + "eval_steps_per_second": 12.388, + "step": 450 + }, + { + "epoch": 0.6201550387596899, + "grad_norm": 21.875, + "learning_rate": 0.0005, + "loss": 53.53094482421875, + "step": 460 + }, + { + "epoch": 0.647118301314459, + "grad_norm": 20.625, + "learning_rate": 0.0005, + "loss": 53.14410400390625, + "step": 480 + }, + { + "epoch": 0.6740815638692281, + "grad_norm": 19.75, + "learning_rate": 0.0005, + "loss": 52.86866455078125, + "step": 500 + }, + { + "epoch": 0.6740815638692281, + "eval_loss": 3.2967636585235596, + "eval_runtime": 9.5934, + "eval_samples_per_second": 993.08, + "eval_steps_per_second": 12.509, + "step": 500 + }, + { + "epoch": 0.7010448264239973, + "grad_norm": 31.0, + "learning_rate": 0.0005, + "loss": 52.4890625, + "step": 520 + }, + { + "epoch": 0.7280080889787665, + "grad_norm": 16.875, + "learning_rate": 0.0005, + "loss": 52.321435546875, + "step": 540 + }, + { + "epoch": 0.741489720256151, + "eval_loss": 3.250918388366699, + "eval_runtime": 9.1469, + "eval_samples_per_second": 1041.558, + "eval_steps_per_second": 13.119, + "step": 550 + }, + { + "epoch": 0.7549713515335356, + "grad_norm": 18.5, + "learning_rate": 0.0005, + "loss": 52.020050048828125, + "step": 560 + }, + { + "epoch": 0.7819346140883047, + "grad_norm": 23.875, + "learning_rate": 0.0005, + "loss": 51.72322387695313, + "step": 580 + }, + { + "epoch": 0.8088978766430738, + "grad_norm": 17.125, + "learning_rate": 0.0005, + "loss": 51.40700073242188, + "step": 600 + }, + { + "epoch": 0.8088978766430738, + "eval_loss": 3.21195650100708, + "eval_runtime": 9.4995, + "eval_samples_per_second": 1002.9, + "eval_steps_per_second": 12.632, + "step": 600 + }, + { + "epoch": 0.8358611391978429, + "grad_norm": 18.125, + "learning_rate": 0.0005, + "loss": 51.1299072265625, + "step": 620 + }, + { + "epoch": 0.8628244017526121, + "grad_norm": 18.875, + "learning_rate": 0.0005, + "loss": 50.86155700683594, + "step": 640 + }, + { + "epoch": 0.8763060330299967, + "eval_loss": 3.1778173446655273, + "eval_runtime": 9.5123, + "eval_samples_per_second": 1001.54, + "eval_steps_per_second": 12.615, + "step": 650 + }, + { + "epoch": 0.8897876643073812, + "grad_norm": 30.125, + "learning_rate": 0.0005, + "loss": 50.78846435546875, + "step": 660 + }, + { + "epoch": 0.9167509268621503, + "grad_norm": 23.0, + "learning_rate": 0.0005, + "loss": 50.50668640136719, + "step": 680 + }, + { + "epoch": 0.9437141894169194, + "grad_norm": 18.25, + "learning_rate": 0.0005, + "loss": 50.196780395507815, + "step": 700 + }, + { + "epoch": 0.9437141894169194, + "eval_loss": 3.135035991668701, + "eval_runtime": 9.3872, + "eval_samples_per_second": 1014.894, + "eval_steps_per_second": 12.783, + "step": 700 + }, + { + "epoch": 0.9706774519716885, + "grad_norm": 21.75, + "learning_rate": 0.0005, + "loss": 49.94974365234375, + "step": 720 + }, + { + "epoch": 0.9976407145264578, + "grad_norm": 25.125, + "learning_rate": 0.0005, + "loss": 49.74850769042969, + "step": 740 + }, + { + "epoch": 1.0107853050219076, + "eval_loss": 3.108103036880493, + "eval_runtime": 9.4956, + "eval_samples_per_second": 1003.31, + "eval_steps_per_second": 12.637, + "step": 750 + } + ], + "logging_steps": 20, + "max_steps": 750, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 100, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": true + }, + "attributes": {} + } + }, + "total_flos": 4354347480907776.0, + "train_batch_size": 80, + "trial_name": null, + "trial_params": null +} diff --git a/outio/mlp-linear-9L_run/checkpoint-750/training_args.bin b/outio/mlp-linear-9L_run/checkpoint-750/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..b3fbb54ac1ee27b6a4fde07a6a6bd316742ee7a5 --- /dev/null +++ b/outio/mlp-linear-9L_run/checkpoint-750/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fa9985a7c7f521fa2a6f7d16c7d254b6d5e243707cd393fa6ea6dfe9b9926147 +size 4920 diff --git a/outio/mlp-linear-9L_run/config.json b/outio/mlp-linear-9L_run/config.json new file mode 100644 index 0000000000000000000000000000000000000000..2c28744029e077a40cfd0de09e0957560eb5ab00 --- /dev/null +++ b/outio/mlp-linear-9L_run/config.json @@ -0,0 +1,35 @@ +{ + "activation": "linear", + "architectures": [ + "TinyLlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 1, + "dtype": "bfloat16", + "eos_token_id": 2, + "head_dim": 32, + "hidden_act": "silu", + "hidden_size": 128, + "initializer_range": 0.02, + "intermediate_size": 256, + "max_position_embeddings": 512, + "mlp_bias": false, + "mlp_type": "mlp", + "model_type": "tiny_llama", + "num_attention_heads": 4, + "num_hidden_layers": 9, + "num_key_value_heads": 4, + "pad_token_id": 0, + "pretraining_tp": 1, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "rope_theta": 10000.0, + "rope_type": "default" + }, + "tie_word_embeddings": true, + "tokenizer_name": "w-ahmad/tiny-stories-tokenizer", + "transformers_version": "5.15.0.dev0", + "use_cache": false, + "vocab_size": 4096 +} diff --git a/outio/mlp-linear-9L_run/model.safetensors b/outio/mlp-linear-9L_run/model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..ffd8224aa84343ad75a0ed106051be614a4dca07 --- /dev/null +++ b/outio/mlp-linear-9L_run/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0b5e551dcb45402eb46f36dadcb4bb5f569e4a98cac7b8ef431d7311497cad45 +size 4010544 diff --git a/outio/mlp-linear-9L_run/tokenizer.json b/outio/mlp-linear-9L_run/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..4e4e61f2579f8f18013caa72cd48f66b9ed1ad9f --- /dev/null +++ b/outio/mlp-linear-9L_run/tokenizer.json @@ -0,0 +1,19501 @@ +{ + "version": "1.0", + "truncation": null, + "padding": null, + "added_tokens": [ + { + "id": 0, + "content": "<|endoftext|>", + "single_word": false, + "lstrip": false, + "rstrip": false, + "normalized": true, + "special": true + } + ], + "normalizer": null, + "pre_tokenizer": { + "type": "ByteLevel", + "add_prefix_space": false, + "trim_offsets": true, + "use_regex": true + }, + "post_processor": { + "type": "ByteLevel", + "add_prefix_space": true, + "trim_offsets": false, + "use_regex": true + }, + "decoder": { + "type": "ByteLevel", + "add_prefix_space": true, + "trim_offsets": true, + "use_regex": true + }, + "model": { + "type": "BPE", + "dropout": null, + "unk_token": null, + "continuing_subword_prefix": "", + "end_of_word_suffix": "", + "fuse_unk": false, + "byte_fallback": false, + "ignore_merges": false, + "vocab": { + "<|endoftext|>": 0, + "!": 1, + "\"": 2, + "#": 3, + "$": 4, + "%": 5, + "&": 6, + "'": 7, + "(": 8, + ")": 9, + "*": 10, + "+": 11, + ",": 12, + "-": 13, + ".": 14, + "/": 15, + "0": 16, + "1": 17, + "2": 18, + "3": 19, + "4": 20, + "5": 21, + "6": 22, + "7": 23, + "8": 24, + "9": 25, + ":": 26, + ";": 27, + "<": 28, + "=": 29, + ">": 30, + "?": 31, + "@": 32, + "A": 33, + "B": 34, + "C": 35, + "D": 36, + "E": 37, + "F": 38, + "G": 39, + "H": 40, + "I": 41, + "J": 42, + "K": 43, + "L": 44, + "M": 45, + "N": 46, + "O": 47, + "P": 48, + "Q": 49, + "R": 50, + "S": 51, + "T": 52, + "U": 53, + "V": 54, + "W": 55, + "X": 56, + "Y": 57, + "Z": 58, + "[": 59, + "\\": 60, + "]": 61, + "^": 62, + "_": 63, + "`": 64, + "a": 65, + "b": 66, + "c": 67, + "d": 68, + "e": 69, + "f": 70, + "g": 71, + "h": 72, + "i": 73, + "j": 74, + "k": 75, + "l": 76, + "m": 77, + "n": 78, + "o": 79, + "p": 80, + "q": 81, + "r": 82, + "s": 83, + "t": 84, + "u": 85, + "v": 86, + "w": 87, + "x": 88, + "y": 89, + "z": 90, + "{": 91, + "|": 92, + "}": 93, + "~": 94, + "¡": 95, + "¢": 96, + "£": 97, + "¤": 98, + "¥": 99, + "¦": 100, + "§": 101, + "¨": 102, + "©": 103, + "ª": 104, + "«": 105, + "¬": 106, + "®": 107, + "¯": 108, + "°": 109, + "±": 110, + "²": 111, + "³": 112, + "´": 113, + "µ": 114, + "¶": 115, + "·": 116, + "¸": 117, + "¹": 118, + "º": 119, + "»": 120, + "¼": 121, + "½": 122, + "¾": 123, + "¿": 124, + "À": 125, + "Á": 126, + "Â": 127, + "Ã": 128, + "Ä": 129, + "Å": 130, + "Æ": 131, + "Ç": 132, + "È": 133, + "É": 134, + "Ê": 135, + "Ë": 136, + "Ì": 137, + "Í": 138, + "Î": 139, + "Ï": 140, + "Ð": 141, + "Ñ": 142, + "Ò": 143, + "Ó": 144, + "Ô": 145, + "Õ": 146, + "Ö": 147, + "×": 148, + "Ø": 149, + "Ù": 150, + "Ú": 151, + "Û": 152, + "Ü": 153, + "Ý": 154, + "Þ": 155, + "ß": 156, + "à": 157, + "á": 158, + "â": 159, + "ã": 160, + "ä": 161, + "å": 162, + "æ": 163, + "ç": 164, + "è": 165, + "é": 166, + "ê": 167, + "ë": 168, + "ì": 169, + "í": 170, + "î": 171, + "ï": 172, + "ð": 173, + "ñ": 174, + "ò": 175, + "ó": 176, + "ô": 177, + "õ": 178, + "ö": 179, + "÷": 180, + "ø": 181, + "ù": 182, + "ú": 183, + "û": 184, + "ü": 185, + "ý": 186, + "þ": 187, + "ÿ": 188, + "Ā": 189, + "ā": 190, + "Ă": 191, + "ă": 192, + "Ą": 193, + "ą": 194, + "Ć": 195, + "ć": 196, + "Ĉ": 197, + "ĉ": 198, + "Ċ": 199, + "ċ": 200, + "Č": 201, + "č": 202, + "Ď": 203, + "ď": 204, + "Đ": 205, + "đ": 206, + "Ē": 207, + "ē": 208, + "Ĕ": 209, + "ĕ": 210, + "Ė": 211, + "ė": 212, + "Ę": 213, + "ę": 214, + "Ě": 215, + "ě": 216, + "Ĝ": 217, + "ĝ": 218, + "Ğ": 219, + "ğ": 220, + "Ġ": 221, + "ġ": 222, + "Ģ": 223, + "ģ": 224, + "Ĥ": 225, + "ĥ": 226, + "Ħ": 227, + "ħ": 228, + "Ĩ": 229, + "ĩ": 230, + "Ī": 231, + "ī": 232, + "Ĭ": 233, + "ĭ": 234, + "Į": 235, + "į": 236, + "İ": 237, + "ı": 238, + "IJ": 239, + "ij": 240, + "Ĵ": 241, + "ĵ": 242, + "Ķ": 243, + "ķ": 244, + "ĸ": 245, + "Ĺ": 246, + "ĺ": 247, + "Ļ": 248, + "ļ": 249, + "Ľ": 250, + "ľ": 251, + "Ŀ": 252, + "ŀ": 253, + "Ł": 254, + "ł": 255, + "Ń": 256, + "he": 257, + "Ġt": 258, + "Ġa": 259, + "Ġs": 260, + "Ġw": 261, + "nd": 262, + "Ġthe": 263, + "ed": 264, + "Ġand": 265, + "Ġto": 266, + "Ġb": 267, + "in": 268, + "Ġh": 269, + "Ġwa": 270, + "re": 271, + "Ġf": 272, + "it": 273, + "ou": 274, + "Ġc": 275, + "Ġl": 276, + "Ġhe": 277, + "Ġd": 278, + "er": 279, + "Ġwas": 280, + "Ġm": 281, + "Ġp": 282, + "om": 283, + "ĠT": 284, + "Ġo": 285, + "ay": 286, + "ar": 287, + "ing": 288, + "is": 289, + "Ġg": 290, + "il": 291, + "id": 292, + "at": 293, + "en": 294, + "Ġn": 295, + "Ġsa": 296, + "Ġha": 297, + "ĠS": 298, + "im": 299, + "an": 300, + "ĠThe": 301, + "or": 302, + "on": 303, + "Ġit": 304, + "Ġth": 305, + "ll": 306, + "le": 307, + "ĠH": 308, + "Ġher": 309, + "et": 310, + "ot": 311, + "ir": 312, + "ĠShe": 313, + "ĠHe": 314, + "ver": 315, + "es": 316, + "Ġin": 317, + "ut": 318, + "ow": 319, + "ck": 320, + "Ġe": 321, + "Ġu": 322, + "ld": 323, + "ĠThey": 324, + "oo": 325, + "ig": 326, + "Ġsaid": 327, + "am": 328, + "ily": 329, + "Ġbe": 330, + "Ġy": 331, + "Ġr": 332, + "Ġst": 333, + "ce": 334, + "Ġshe": 335, + "Ġ\"": 336, + "pp": 337, + "ke": 338, + "ith": 339, + "On": 340, + "ĠI": 341, + "Ġwith": 342, + "ve": 343, + "Lily": 344, + "Ġon": 345, + "Ġof": 346, + "Ġso": 347, + "Ġhis": 348, + "ked": 349, + "ri": 350, + "nt": 351, + "very": 352, + "Ġpl": 353, + "Ġday": 354, + "ad": 355, + "Ġyou": 356, + "Ġthat": 357, + "Ġup": 358, + "Ġhad": 359, + "st": 360, + "Ġplay": 361, + "Ġthey": 362, + "ĠLily": 363, + "Ġwe": 364, + "Ġmom": 365, + "my": 366, + "Ġfor": 367, + "el": 368, + "ould": 369, + "un": 370, + "ĠB": 371, + "'s": 372, + "itt": 373, + "ent": 374, + "Ġhapp": 375, + "The": 376, + "ch": 377, + "Ġli": 378, + "out": 379, + "Ġwant": 380, + "Ġsh": 381, + "her": 382, + "ly": 383, + "ime": 384, + "ittle": 385, + "ound": 386, + "Ġvery": 387, + "Ġtime": 388, + "ome": 389, + "Ġlittle": 390, + "Ġthere": 391, + "se": 392, + "Ġdo": 393, + "Ġwh": 394, + "all": 395, + "Ġk": 396, + "end": 397, + "al": 398, + "ht": 399, + "Ġne": 400, + "Ġre": 401, + "Ġnot": 402, + "Ġhappy": 403, + "ĠĊ": 404, + "Ġbig": 405, + "ĠM": 406, + "Ġbut": 407, + "Ġsm": 408, + "ack": 409, + "Ġsaw": 410, + "ĠIt": 411, + "Ġas": 412, + "Ġan": 413, + "ra": 414, + "riend": 415, + "Ġfriend": 416, + "ide": 417, + "One": 418, + "ry": 419, + "'t": 420, + "ved": 421, + "Ġis": 422, + "Once": 423, + "ake": 424, + ".\"": 425, + "Ġwere": 426, + "ter": 427, + "ug": 428, + "Ġloo": 429, + "Ġlo": 430, + "ore": 431, + "ec": 432, + "ĠTim": 433, + "Ġhim": 434, + "Ġbo": 435, + "!\"": 436, + "Ġtoo": 437, + "Ġgo": 438, + "Ġupon": 439, + "irl": 440, + "Ġj": 441, + "Ġwanted": 442, + "Ġgirl": 443, + "Ġse": 444, + "Ġout": 445, + "ard": 446, + "way": 447, + "Ġsp": 448, + "ill": 449, + "ind": 450, + "Ġthem": 451, + "Ġcould": 452, + "fu": 453, + "hen": 454, + "Ġat": 455, + "ur": 456, + "Ġdid": 457, + "Ġsmil": 458, + "Ġtheir": 459, + "Ġare": 460, + "Ġex": 461, + "ĠA": 462, + "ain": 463, + "Ġwent": 464, + "art": 465, + "hed": 466, + "rom": 467, + "ic": 468, + "round": 469, + "Ġhave": 470, + "Ġnam": 471, + "lp": 472, + "Ġall": 473, + "ĠJ": 474, + "ful": 475, + "Ġkn": 476, + "hing": 477, + "ood": 478, + "Ġhelp": 479, + "ight": 480, + "Ġfriends": 481, + "one": 482, + "ark": 483, + "Ġback": 484, + "um": 485, + "Ġcan": 486, + "Ġnamed": 487, + "Ġcl": 488, + "?\"": 489, + "Ġfun": 490, + "are": 491, + "ĠBen": 492, + "Ġloved": 493, + "Ġal": 494, + "elt": 495, + "ĠTimmy": 496, + "ĠOne": 497, + "op": 498, + "side": 499, + "Ġle": 500, + "Ġno": 501, + "Ġsc": 502, + "ĠTom": 503, + "Ġfelt": 504, + "oug": 505, + "Ġsmiled": 506, + "ick": 507, + "Ġasked": 508, + "You": 509, + "Ġtoy": 510, + "Ġman": 511, + "Ġaround": 512, + "ame": 513, + "Ġfe": 514, + "Ġsay": 515, + "Ġboy": 516, + "Ġsome": 517, + "Ġlooked": 518, + "ure": 519, + "omet": 520, + "Ġbr": 521, + "Ġwould": 522, + "Ġme": 523, + "Ġbir": 524, + "Ġlike": 525, + "get": 526, + "Ġstart": 527, + "Ġro": 528, + "as": 529, + "Ġsee": 530, + "ĠW": 531, + "ice": 532, + "ong": 533, + "Ġbird": 534, + "Ġsomet": 535, + "dd": 536, + "Ġwor": 537, + "ade": 538, + "ie": 539, + "king": 540, + "Ġag": 541, + "own": 542, + "Ġtre": 543, + "Ġfa": 544, + "Ġaway": 545, + "Ġwhat": 546, + "ings": 547, + "Ġstarted": 548, + "gether": 549, + "Ġran": 550, + "â": 551, + "âĤ": 552, + "âĤ¬": 553, + "ared": 554, + "Ġmake": 555, + "ĠBut": 556, + "ited": 557, + "if": 558, + "oud": 559, + "Ġmade": 560, + "Ġtogether": 561, + "Ġsomething": 562, + "Ġexc": 563, + "ag": 564, + "Ġco": 565, + "Ġpark": 566, + "Ġnew": 567, + "Ġsad": 568, + "Ġput": 569, + ",\"": 570, + "Ġfrom": 571, + "ble": 572, + "ther": 573, + "Ġpr": 574, + "Ġmu": 575, + "Ġcar": 576, + "Ġhome": 577, + "ĠYou": 578, + "Ġthen": 579, + "Ġwhen": 580, + "Ġfound": 581, + "ell": 582, + "Ġother": 583, + "Ġagain": 584, + "Ġch": 585, + "Ġdec": 586, + "Ġwho": 587, + "Ġla": 588, + "ried": 589, + "ss": 590, + "Ġgood": 591, + "Ġhug": 592, + "ĠL": 593, + "pped": 594, + "Ġwal": 595, + "ep": 596, + "ally": 597, + "Ġsays": 598, + "Ġfl": 599, + "est": 600, + "ach": 601, + "ĠE": 602, + "Ġexcited": 603, + "pl": 604, + "qu": 605, + "ook": 606, + "Ġget": 607, + "ought": 608, + "Ġplaying": 609, + "Ġgot": 610, + "Ġsw": 611, + "ous": 612, + "hat": 613, + "ny": 614, + "ided": 615, + "uck": 616, + "Ġthings": 617, + "Ġevery": 618, + "Ġdecided": 619, + "Ġcame": 620, + "Ġbec": 621, + "ave": 622, + "ro": 623, + "ax": 624, + "Ġliked": 625, + "Ġdown": 626, + "Ġdog": 627, + "Ġscared": 628, + "Ġv": 629, + "udd": 630, + "ust": 631, + "Ġone": 632, + "Ġfind": 633, + "Ġbl": 634, + "Ġthan": 635, + "ĠD": 636, + "ouse": 637, + "ways": 638, + "Ġkne": 639, + "Ġdidn": 640, + "ap": 641, + "Ġmommy": 642, + "Ġcare": 643, + "Ġalways": 644, + "Ġab": 645, + "ist": 646, + "Ġdad": 647, + "Ġfeel": 648, + "ara": 649, + "bb": 650, + "arn": 651, + "fe": 652, + "Ġyour": 653, + "Ġoutside": 654, + "ue": 655, + "Ġgra": 656, + "ant": 657, + "Ġke": 658, + "ĠMom": 659, + "Ġtook": 660, + "Ġlot": 661, + "nn": 662, + "Ġbu": 663, + "Ġabout": 664, + "ess": 665, + "ĠF": 666, + "nder": 667, + "Ġtree": 668, + "eci": 669, + "Ġlook": 670, + "Ġpo": 671, + "Ġmy": 672, + "our": 673, + "Ġtoys": 674, + "ite": 675, + "ched": 676, + "Ġknew": 677, + "Ġthought": 678, + "ened": 679, + "Ġlearn": 680, + "Ġint": 681, + "Ġold": 682, + "Ġmore": 683, + "nna": 684, + "ise": 685, + "ged": 686, + "Ġta": 687, + "udden": 688, + "ecial": 689, + "Ġspecial": 690, + "ĠMax": 691, + "au": 692, + "Ġwill": 693, + "ers": 694, + "They": 695, + "ret": 696, + "Ġpe": 697, + "Ġho": 698, + "ĠSam": 699, + "Ġtake": 700, + "Ġball": 701, + "Ġknow": 702, + "Ġlaug": 703, + "fter": 704, + "uddenly": 705, + "Ġcat": 706, + "Ġhow": 707, + "ive": 708, + "Ġtr": 709, + "Ġmuch": 710, + "Ġany": 711, + "Ġpu": 712, + "ma": 713, + "Ġsl": 714, + "Ġsor": 715, + "Ġmo": 716, + "ven": 717, + "ish": 718, + "Ġshow": 719, + "Ġcouldn": 720, + "ause": 721, + "ink": 722, + "But": 723, + "Ġhouse": 724, + "Ġinto": 725, + "ump": 726, + "Ġover": 727, + "Ġtried": 728, + "and": 729, + "Ġeat": 730, + "Ġsk": 731, + "Ġsun": 732, + "Ġtw": 733, + "Ġclo": 734, + "ia": 735, + "Ġrun": 736, + "Ġhand": 737, + "ĠEvery": 738, + "Ġen": 739, + "Ġinside": 740, + "dy": 741, + "Ġif": 742, + "Ġtold": 743, + "Ġnever": 744, + "ion": 745, + "Ġqu": 746, + "Ġbecause": 747, + "by": 748, + "Ġproud": 749, + "Ġgave": 750, + "Ġsorry": 751, + "Ġthis": 752, + "Ġop": 753, + "Ġplayed": 754, + "ate": 755, + "Ġexpl": 756, + "Ġheard": 757, + "ty": 758, + "Ġor": 759, + "Ġwater": 760, + "ank": 761, + "ge": 762, + "sed": 763, + "Ġpick": 764, + "other": 765, + "Ġroom": 766, + "etter": 767, + "Ġjust": 768, + "ace": 769, + "Ġhugged": 770, + "ak": 771, + "Ġgre": 772, + "here": 773, + "Ġoff": 774, + "ĠSara": 775, + "Ġpret": 776, + "ile": 777, + "Ġeach": 778, + "Ġcom": 779, + "Ġlong": 780, + "Ġbox": 781, + "ort": 782, + "Ġstr": 783, + "iz": 784, + "Ġunt": 785, + "Ġwat": 786, + "oth": 787, + "Ġneed": 788, + "Ġjo": 789, + "ĠWe": 790, + "Tom": 791, + "Ġsmall": 792, + "ine": 793, + "Ġbear": 794, + "Mom": 795, + "Ġuntil": 796, + "Ġnice": 797, + "Ġtry": 798, + "ving": 799, + "uc": 800, + "sel": 801, + "ough": 802, + "Ġlearned": 803, + "Ġkind": 804, + "ĠAnna": 805, + "ild": 806, + "Ġfo": 807, + "Ġmany": 808, + "'m": 809, + "ĠJack": 810, + "Ġbetter": 811, + "Ġim": 812, + "gry": 813, + "imal": 814, + "Ġanimal": 815, + "urt": 816, + "ft": 817, + "Ġend": 818, + "Ġsn": 819, + "aut": 820, + "Ġte": 821, + "Ġcle": 822, + "ĠJo": 823, + "vent": 824, + "urp": 825, + "Ġgr": 826, + "Ġbeaut": 827, + "Ġjump": 828, + "mb": 829, + "Ġad": 830, + "ream": 831, + "pt": 832, + "ĠSo": 833, + "ĠHer": 834, + "Ġflow": 835, + "ies": 836, + "Ġche": 837, + "Ġbra": 838, + "Ġthanked": 839, + "Ġeven": 840, + "Ġbest": 841, + "Ġcall": 842, + "ady": 843, + "He": 844, + "Ġlots": 845, + "Ġlaughed": 846, + "self": 847, + "Ġra": 848, + "Ġway": 849, + "ars": 850, + "urn": 851, + "Ġby": 852, + "Ġfast": 853, + "lly": 854, + "Ġfam": 855, + "ĠC": 856, + "ves": 857, + "arden": 858, + "Ġgarden": 859, + "Ġbeauti": 860, + "wn": 861, + "Th": 862, + "Ġbeautiful": 863, + "Ġloud": 864, + "lew": 865, + "Ġsky": 866, + "Ġdon": 867, + "hn": 868, + "ered": 869, + "iny": 870, + "Ġcareful": 871, + "Ġlove": 872, + "Ġfi": 873, + "ĠThen": 874, + "Åĵ": 875, + "ase": 876, + "ect": 877, + "Ġsafe": 878, + "ĠAnd": 879, + "Ġunder": 880, + "Ġcome": 881, + "ĠFrom": 882, + "Yes": 883, + "ĠMia": 884, + "It": 885, + "me": 886, + "Ġhard": 887, + "Ġcu": 888, + "Ġwo": 889, + "Ġlist": 890, + "Ġstay": 891, + "ane": 892, + "sh": 893, + "ople": 894, + "Ġgl": 895, + "ning": 896, + "Ġstill": 897, + "ool": 898, + "Ġhurt": 899, + "ree": 900, + "ĠHis": 901, + "Ġimp": 902, + "Ġfamily": 903, + "Ġâ": 904, + "Ġboth": 905, + "rm": 906, + "igh": 907, + "Ġlived": 908, + "hes": 909, + "When": 910, + "Ġpeople": 911, + "Ġanimals": 912, + "Ġcol": 913, + "Ġbrave": 914, + "Ġwalked": 915, + "ob": 916, + "Tim": 917, + "ct": 918, + "Ġlet": 919, + "urpr": 920, + "ĠWhen": 921, + "Ġtwo": 922, + "Ġsurpr": 923, + "Ġshould": 924, + "ished": 925, + "Ġbad": 926, + "ress": 927, + "Ġkept": 928, + "Ġfore": 929, + "Ġflew": 930, + "Ġfin": 931, + "Ġstor": 932, + "Ġfly": 933, + "ast": 934, + "ised": 935, + "ip": 936, + "Ġits": 937, + "led": 938, + "ock": 939, + "ucy": 940, + "fore": 941, + "Ġgoing": 942, + "Ġclean": 943, + "Ġdan": 944, + "Ġpic": 945, + "Ġsoon": 946, + "Ġcalled": 947, + "Ġshare": 948, + "kay": 949, + "Ġangry": 950, + "Ġrock": 951, + "Ġcon": 952, + "Ġpretty": 953, + "No": 954, + "Ġide": 955, + "ied": 956, + "illy": 957, + "Ġground": 958, + "xt": 959, + "Ġred": 960, + "Ġexplore": 961, + "Ġcry": 962, + "Ġadvent": 963, + "Ġsto": 964, + "so": 965, + "Ġreal": 966, + "Let": 967, + "Ġwind": 968, + "Ġshiny": 969, + "be": 970, + "Ġbook": 971, + "Ġalso": 972, + "Ġdoll": 973, + "Ġidea": 974, + "Ġbefore": 975, + "Ġopened": 976, + "dded": 977, + "Ġwhile": 978, + "ummy": 979, + "Ġkeep": 980, + "Ġey": 981, + "Ġnow": 982, + "Ġdoor": 983, + "Ġfeeling": 984, + "âĤ¬â": 985, + "oon": 986, + "oy": 987, + "Ġwalking": 988, + "Ġnoise": 989, + "Ġfr": 990, + "les": 991, + "age": 992, + "ious": 993, + "Ġcolor": 994, + "Ġturn": 995, + "thing": 996, + "ff": 997, + "uch": 998, + "th": 999, + "Ġbed": 1000, + "ary": 1001, + "Ġdra": 1002, + "Ġpicked": 1003, + "imb": 1004, + "eet": 1005, + "Ġclimb": 1006, + "Ġdel": 1007, + "What": 1008, + "Ġbeing": 1009, + "Ġfood": 1010, + "Ġun": 1011, + "Ġfar": 1012, + "ture": 1013, + "joy": 1014, + "Ġadventure": 1015, + "ac": 1016, + "Ġsmile": 1017, + "Ġdif": 1018, + "Ħ¢": 1019, + "âĤ¬âĦ¢": 1020, + "memb": 1021, + "Ġwr": 1022, + "Ġthr": 1023, + "ught": 1024, + "Ġlooking": 1025, + "Ġnext": 1026, + "iced": 1027, + "ĠLucy": 1028, + "Ġnodded": 1029, + "Ġquick": 1030, + "ĠP": 1031, + "Ġdis": 1032, + "Ġrepl": 1033, + "ĠDad": 1034, + "Ġwait": 1035, + "ized": 1036, + "Ġforest": 1037, + "Ġclos": 1038, + "ĠSuddenly": 1039, + "Ġtra": 1040, + "That": 1041, + "Ġeyes": 1042, + "ger": 1043, + "bbit": 1044, + "ted": 1045, + "Ġown": 1046, + "Ġrain": 1047, + "Ġimport": 1048, + "Ġgreat": 1049, + "Ġrememb": 1050, + "Ġpicture": 1051, + "Thank": 1052, + "Suddenly": 1053, + "Ġstopped": 1054, + "Ġenjoy": 1055, + "Ġvo": 1056, + "Ben": 1057, + "Ġgive": 1058, + "Ġimportant": 1059, + "Ġwork": 1060, + "Ġnear": 1061, + "gan": 1062, + "pot": 1063, + "Ġever": 1064, + "Ġapp": 1065, + "Ġafter": 1066, + "Ġquickly": 1067, + "Ġlisten": 1068, + "Ġbre": 1069, + "ting": 1070, + "bbed": 1071, + "Ġma": 1072, + "Ġfish": 1073, + "Ġreplied": 1074, + "Ġrabbit": 1075, + "Ġhands": 1076, + "ĠG": 1077, + "Ġnoticed": 1078, + "Ġbro": 1079, + "Ġslide": 1080, + "Ġthink": 1081, + "Ġwalk": 1082, + "Ġtruck": 1083, + "Ġac": 1084, + "kes": 1085, + "Ġstrong": 1086, + "She": 1087, + "Ġshowed": 1088, + "Ġde": 1089, + "Ġeveryone": 1090, + "Ġwonder": 1091, + "fere": 1092, + "bye": 1093, + "Ġdiffere": 1094, + "irst": 1095, + "So": 1096, + "Ġsure": 1097, + "Ġhas": 1098, + "Ġright": 1099, + "Ġbeen": 1100, + "Ġbecame": 1101, + "Ġsound": 1102, + "maz": 1103, + "Ġtow": 1104, + "Ġru": 1105, + "Ġtal": 1106, + "Ġamaz": 1107, + "Ġhead": 1108, + "Ġshout": 1109, + "Ġbright": 1110, + "Ġye": 1111, + "Ġwatched": 1112, + "ĠR": 1113, + "Ġmor": 1114, + "Ġchild": 1115, + "able": 1116, + "Ġmean": 1117, + "Ġwhere": 1118, + "llow": 1119, + "Ġhigh": 1120, + "ĠSue": 1121, + "Ġface": 1122, + "Ġcook": 1123, + "day": 1124, + "aybe": 1125, + "Ġwatch": 1126, + "Ġblue": 1127, + "aught": 1128, + "Ġdifferent": 1129, + "Ġstore": 1130, + "ĠN": 1131, + "Ġgoodbye": 1132, + "Ġdress": 1133, + "ull": 1134, + "ĠBob": 1135, + "ng": 1136, + "ange": 1137, + "Ġsqu": 1138, + "Ġokay": 1139, + "isy": 1140, + "lease": 1141, + "ĠMommy": 1142, + "Ġvoice": 1143, + "Jo": 1144, + "ath": 1145, + "Ġnight": 1146, + "ĠSpot": 1147, + "Ġus": 1148, + "Ġboat": 1149, + "Ġflowers": 1150, + "Ġplace": 1151, + "Ġfollow": 1152, + "Ġar": 1153, + "Ġuse": 1154, + "Ġcloser": 1155, + "unny": 1156, + "leep": 1157, + "ired": 1158, + "Ġfav": 1159, + "Ġyell": 1160, + "Ġgrabbed": 1161, + "Ġcuri": 1162, + "Ġwarm": 1163, + "Ġcr": 1164, + "Ġforg": 1165, + "ĠSarah": 1166, + "ĠJohn": 1167, + "Ġmag": 1168, + "Ġstick": 1169, + "We": 1170, + "Ġjumped": 1171, + "Ġcake": 1172, + "more": 1173, + "Ġtell": 1174, + "Ġanymore": 1175, + "After": 1176, + "Ġbutter": 1177, + "ndma": 1178, + "Ġthree": 1179, + "Ġask": 1180, + "co": 1181, + "Ġour": 1182, + "lie": 1183, + "Ġcurious": 1184, + "ount": 1185, + "orn": 1186, + "Ġcont": 1187, + "Ġfell": 1188, + "ached": 1189, + "ĠTh": 1190, + "Ġbirds": 1191, + "ass": 1192, + "Her": 1193, + "iss": 1194, + "Ġhelped": 1195, + "ĠJane": 1196, + "Ġpull": 1197, + "Ġfirst": 1198, + "itc": 1199, + "Ġblock": 1200, + "Ġhop": 1201, + "Ġbit": 1202, + "Ġdr": 1203, + "Ġrealized": 1204, + "Ġkid": 1205, + "Look": 1206, + "ila": 1207, + "Ġmon": 1208, + "Ġbrother": 1209, + "Anna": 1210, + "Sara": 1211, + "Ġz": 1212, + "ĠEveryone": 1213, + "Ġate": 1214, + "Ġdoes": 1215, + "imes": 1216, + "Ġhappened": 1217, + "Ġstop": 1218, + "zy": 1219, + "Ġyummy": 1220, + "Ġfavor": 1221, + "ppy": 1222, + "Ġkitc": 1223, + "Ġkitchen": 1224, + "Ġsweet": 1225, + "us": 1226, + "Ġper": 1227, + "Ġreally": 1228, + "aisy": 1229, + "Ġgrass": 1230, + "Ġfavorite": 1231, + "Ġbegan": 1232, + "Ġrest": 1233, + "Ġready": 1234, + "Ġlea": 1235, + "Ġreached": 1236, + "Ġunderst": 1237, + "air": 1238, + "Ġste": 1239, + "Ġbunny": 1240, + "ĠAs": 1241, + "Ġstory": 1242, + "'re": 1243, + "Ġpain": 1244, + "Ġsing": 1245, + "ster": 1246, + "Ġhere": 1247, + "Ġsand": 1248, + "Ġonly": 1249, + "Ġflo": 1250, + "Ġam": 1251, + "Ġglad": 1252, + "Ġtri": 1253, + "Ġbeh": 1254, + "Ġworld": 1255, + "Ġopen": 1256, + "Ġcre": 1257, + "fully": 1258, + "Ġprin": 1259, + "where": 1260, + "arent": 1261, + "Ġflower": 1262, + "Ġthrough": 1263, + "Ġba": 1264, + "Ġfire": 1265, + "Ġdone": 1266, + "Ġhaving": 1267, + "Ġthing": 1268, + "Ġdelic": 1269, + "Ġhimself": 1270, + "Ġtired": 1271, + "Ġparent": 1272, + "Ġsoft": 1273, + "Ġfro": 1274, + "Timmy": 1275, + "Ġtast": 1276, + "Ġbutterf": 1277, + "ĠLet": 1278, + "Ġcut": 1279, + "Ġpart": 1280, + "Ġwhy": 1281, + "ken": 1282, + "Ġmess": 1283, + "Can": 1284, + "Ġworry": 1285, + "Mommy": 1286, + "iver": 1287, + "Ġdin": 1288, + "uff": 1289, + "ater": 1290, + "Ġmagic": 1291, + "Ġwaved": 1292, + "Ġshouted": 1293, + "Ġpond": 1294, + "Ġkids": 1295, + "Ġhat": 1296, + "Ġduck": 1297, + "Ġsees": 1298, + "olly": 1299, + "illed": 1300, + "Ġgame": 1301, + "ient": 1302, + "Ġmaking": 1303, + "ather": 1304, + "John": 1305, + "As": 1306, + "akes": 1307, + "Ġcatch": 1308, + "Ġseen": 1309, + "Ġcool": 1310, + "ation": 1311, + "Ġcoming": 1312, + "Ġless": 1313, + "Ġdark": 1314, + "Ġused": 1315, + "eddy": 1316, + "Ġfix": 1317, + "ĠJoe": 1318, + "Ġthank": 1319, + "mer": 1320, + "Ġtop": 1321, + "Ġlady": 1322, + "Ġhair": 1323, + "aring": 1324, + "Ġprom": 1325, + "Ġhopped": 1326, + "ĠCan": 1327, + "\".": 1328, + "ign": 1329, + "Ġsurprise": 1330, + "Ġdraw": 1331, + "Ġmum": 1332, + "Ġmouse": 1333, + "Ġfunny": 1334, + "rel": 1335, + "Ġcra": 1336, + "Ġ-": 1337, + "aper": 1338, + "Ġfull": 1339, + "Ġtouch": 1340, + "Ġlight": 1341, + "Ġspot": 1342, + "ren": 1343, + "Ġdro": 1344, + "ĠBenny": 1345, + "Ġparents": 1346, + "Ġworked": 1347, + "ins": 1348, + "oney": 1349, + "ĠDo": 1350, + "Ġsurprised": 1351, + "Ġcarefully": 1352, + "Ġtrees": 1353, + "Ġfrog": 1354, + "Ġswing": 1355, + "Ġdoing": 1356, + "Don": 1357, + "Ġsat": 1358, + "inally": 1359, + "ĠThat": 1360, + "Ġice": 1361, + "ĠIn": 1362, + "Ġheld": 1363, + "Wow": 1364, + "Ġrunning": 1365, + "Ġpretend": 1366, + "Ġswim": 1367, + "Ġset": 1368, + "Ġread": 1369, + "Ġdelicious": 1370, + "Ġneeded": 1371, + "Ġtight": 1372, + "Ġslow": 1373, + "Ġremembered": 1374, + "Ġlost": 1375, + "Ġcold": 1376, + "Ġsmell": 1377, + "ards": 1378, + "Ġwood": 1379, + "Ġhappily": 1380, + "hy": 1381, + "ely": 1382, + "Ġlooks": 1383, + "Ġbehind": 1384, + "Ġherself": 1385, + "Ġcried": 1386, + "Ġenjoyed": 1387, + "Ġname": 1388, + "ask": 1389, + "Ġyears": 1390, + "Ġbuy": 1391, + "Ġhole": 1392, + "Ġblocks": 1393, + "Ġdri": 1394, + "Ġsleep": 1395, + "Ġgi": 1396, + "Ġyellow": 1397, + "cess": 1398, + "Ġbutterfly": 1399, + "ike": 1400, + "ued": 1401, + "Ġwished": 1402, + "Ġperf": 1403, + "Ġanother": 1404, + "ĠDaisy": 1405, + "Ġgreen": 1406, + "Ġmove": 1407, + "Ġtall": 1408, + "Ġair": 1409, + "Ġbag": 1410, + "Ġfloor": 1411, + "Ġcars": 1412, + "Ġlesson": 1413, + "ul": 1414, + "Ġbow": 1415, + "Ġarri": 1416, + "Ġwindow": 1417, + "Ġsho": 1418, + "Ġchildren": 1419, + "ner": 1420, + "Ġfinished": 1421, + "Ġhill": 1422, + "ĠAfter": 1423, + "andy": 1424, + "sp": 1425, + "ens": 1426, + "Ġpaper": 1427, + "Ġlikes": 1428, + "From": 1429, + "sy": 1430, + "Ġhold": 1431, + "Sam": 1432, + "Ġwrong": 1433, + "reed": 1434, + "Ġcontin": 1435, + "Ġunderstand": 1436, + "atter": 1437, + "uit": 1438, + "rew": 1439, + "Ġgent": 1440, + "Ġanything": 1441, + "Ġclose": 1442, + "Ġwall": 1443, + "Ġel": 1444, + "asure": 1445, + "Ġleft": 1446, + "Ġable": 1447, + "Ġarrived": 1448, + "Ġhun": 1449, + "ross": 1450, + "av": 1451, + "Ġhear": 1452, + "Ġfriendly": 1453, + "Ġforgot": 1454, + "Ġgone": 1455, + "ama": 1456, + "Ġcreat": 1457, + "Ġwet": 1458, + "Ġlion": 1459, + "Hell": 1460, + "Hello": 1461, + "Ġdream": 1462, + "Ġhot": 1463, + "bo": 1464, + "ber": 1465, + "ĠSally": 1466, + "Ġfilled": 1467, + "Ġdir": 1468, + "ield": 1469, + "Ġcookies": 1470, + "Ġlaugh": 1471, + "Ġbroken": 1472, + "Ġdinner": 1473, + "lf": 1474, + "Ġelse": 1475, + "Ġhid": 1476, + "ced": 1477, + "Ġpink": 1478, + "Ġfollowed": 1479, + "Ġtable": 1480, + "Ġmar": 1481, + "Ġcolors": 1482, + "oup": 1483, + "Ġmoment": 1484, + "Ġcontinued": 1485, + "Ġwonderful": 1486, + "ĠEm": 1487, + "ey": 1488, + "ined": 1489, + "irrel": 1490, + "rot": 1491, + "Ġfinally": 1492, + "Jack": 1493, + "Ġarm": 1494, + "Ġmight": 1495, + "ĠNow": 1496, + "Ġcast": 1497, + "Ġyes": 1498, + "Ġbuild": 1499, + "Ġenough": 1500, + "ĠBilly": 1501, + "Ġsomeone": 1502, + "Ġperfect": 1503, + "Ġfair": 1504, + "Ġstayed": 1505, + "app": 1506, + "Ġmorning": 1507, + "ĠThere": 1508, + "Ġpuppy": 1509, + "ol": 1510, + "Ġdaddy": 1511, + "Ġtrying": 1512, + "'ll": 1513, + "Ġju": 1514, + "Ġothers": 1515, + "Ġbooks": 1516, + "Ġagreed": 1517, + "Ġmus": 1518, + "Ġsnow": 1519, + "Ġsquirrel": 1520, + "Ġcream": 1521, + "room": 1522, + "Ġgetting": 1523, + "Ġgrandma": 1524, + "ĠLila": 1525, + "Ġleaves": 1526, + "Ġbaby": 1527, + "Ġcount": 1528, + "icy": 1529, + "Ġfew": 1530, + "At": 1531, + "Ġfall": 1532, + "Then": 1533, + "gon": 1534, + "Ġbreak": 1535, + "Ġclimbed": 1536, + "ries": 1537, + "Ġbeach": 1538, + "ash": 1539, + "Ġbug": 1540, + "Ġplease": 1541, + "Ġmother": 1542, + "Ġbrought": 1543, + "Ġtrain": 1544, + "ance": 1545, + "ĠThis": 1546, + "Ġdeep": 1547, + "Ġvis": 1548, + "ated": 1549, + "shed": 1550, + "Ġmet": 1551, + "Ġwasn": 1552, + "Ġsometimes": 1553, + "Ġeverything": 1554, + "ĠTommy": 1555, + "ich": 1556, + "Ġmusic": 1557, + "key": 1558, + "ling": 1559, + "This": 1560, + "os": 1561, + "Ġturned": 1562, + "lc": 1563, + "Ġride": 1564, + "oom": 1565, + "Ġsong": 1566, + "oring": 1567, + "Ġspark": 1568, + "Ġfight": 1569, + "Ġhungry": 1570, + "ons": 1571, + "Ġpictures": 1572, + "ds": 1573, + "Ġmakes": 1574, + "Ġsear": 1575, + "ucky": 1576, + "Ġlistened": 1577, + "Ġjoy": 1578, + "Ġpol": 1579, + "ĠWhat": 1580, + "His": 1581, + "ee": 1582, + "Ġclot": 1583, + "Ġsail": 1584, + "apped": 1585, + "bble": 1586, + "Ġcomp": 1587, + "Ġplan": 1588, + "Ġwants": 1589, + "Ġclothes": 1590, + "Ġpoin": 1591, + "ĠK": 1592, + "zz": 1593, + "Ġballoon": 1594, + "Ġstrange": 1595, + "man": 1596, + "enny": 1597, + "Ġpi": 1598, + "Ġgrow": 1599, + "Ġflying": 1600, + "row": 1601, + "Ġscary": 1602, + "Ġtreat": 1603, + "Ġwaited": 1604, + "Ġamazed": 1605, + "Ġteddy": 1606, + "Ġcastle": 1607, + "ister": 1608, + "Ġele": 1609, + "Ġamazing": 1610, + "med": 1611, + "Ġspl": 1612, + "Oh": 1613, + "Ġleave": 1614, + "Ġowner": 1615, + "Ġpromised": 1616, + "Ġdanger": 1617, + "ever": 1618, + "OK": 1619, + "Ġsuch": 1620, + "Ġwhite": 1621, + "Ġpat": 1622, + "Ġtalk": 1623, + "uffy": 1624, + "Ġfur": 1625, + "Ġpie": 1626, + "ope": 1627, + "red": 1628, + "Ġpulled": 1629, + "Ġpre": 1630, + "Ġwoods": 1631, + "ĠTo": 1632, + "Ġshared": 1633, + "Ġalone": 1634, + "Ġtowards": 1635, + "cle": 1636, + "ĠMaybe": 1637, + "Ġsit": 1638, + "Ġnap": 1639, + "Ġcolorful": 1640, + "amp": 1641, + "hone": 1642, + "Ġvisit": 1643, + "igg": 1644, + "ugg": 1645, + "thy": 1646, + "ence": 1647, + "Ġtail": 1648, + "ctor": 1649, + "Ġmagical": 1650, + "Ġriver": 1651, + "Ġhig": 1652, + "Ġbelie": 1653, + "Ġgrate": 1654, + "ually": 1655, + "ang": 1656, + "ott": 1657, + "Ġhide": 1658, + "Ġsilly": 1659, + "Ġsmiles": 1660, + "ower": 1661, + "Ġside": 1662, + "Max": 1663, + "Ġroll": 1664, + "Ġgu": 1665, + "ĠAmy": 1666, + "Ġpro": 1667, + "eter": 1668, + "Ġfaster": 1669, + "Ġgrateful": 1670, + "Ġtreasure": 1671, + "ased": 1672, + "Ġsnack": 1673, + "up": 1674, + "Ġland": 1675, + "avy": 1676, + "Ġmoral": 1677, + "ained": 1678, + "Ġsister": 1679, + "ĠGra": 1680, + "Ġshap": 1681, + "Ġpaint": 1682, + "Ġslowly": 1683, + "Ġfield": 1684, + "Ġpet": 1685, + "Ġsec": 1686, + "orm": 1687, + "ĠJim": 1688, + "Ġbran": 1689, + "Ġmin": 1690, + "Ġmust": 1691, + "Ġhelping": 1692, + "Ġworried": 1693, + "Ġjob": 1694, + "Ġfox": 1695, + "set": 1696, + "Ġmoney": 1697, + "Ġpot": 1698, + "Ġqui": 1699, + "Ġbel": 1700, + "Ġacc": 1701, + "Ġliving": 1702, + "Ġmonster": 1703, + "less": 1704, + "Ġparty": 1705, + "ear": 1706, + "Ġmost": 1707, + "Ġwear": 1708, + "Ġremember": 1709, + "ire": 1710, + "Ġsign": 1711, + "wl": 1712, + "Ġcloud": 1713, + "Ġcandy": 1714, + "Ġhigher": 1715, + "ipped": 1716, + "ident": 1717, + "ĠAll": 1718, + "der": 1719, + "ze": 1720, + "Ġquiet": 1721, + "Ġtown": 1722, + "unch": 1723, + "Ġprincess": 1724, + "Ġruns": 1725, + "asket": 1726, + "em": 1727, + "Ġeverywhere": 1728, + "Ġjoin": 1729, + "Ġstepped": 1730, + "Ġbasket": 1731, + "Ġlake": 1732, + "Ġcup": 1733, + "sw": 1734, + "outh": 1735, + "Ġcarrot": 1736, + "Ġcoll": 1737, + "Ġplant": 1738, + "Ġscream": 1739, + "ĠDaddy": 1740, + "Ġbuck": 1741, + "Ġheavy": 1742, + "Ġbigger": 1743, + "Ġcho": 1744, + "Ġblank": 1745, + "Ġclosed": 1746, + "Ġwon": 1747, + "ises": 1748, + "Ġansw": 1749, + "br": 1750, + "Ġblanket": 1751, + "Ġmouth": 1752, + "ames": 1753, + "eces": 1754, + "Ġeating": 1755, + "Ġpieces": 1756, + "cket": 1757, + "Ġstre": 1758, + "Ġwelc": 1759, + "Ġwouldn": 1760, + "Ġreach": 1761, + "Ġbroke": 1762, + "eared": 1763, + "Ġdangerous": 1764, + "gs": 1765, + "ph": 1766, + "ĠMolly": 1767, + "Ġce": 1768, + "ĠMr": 1769, + "ord": 1770, + "fort": 1771, + "Ġdolls": 1772, + "ming": 1773, + "Ġdance": 1774, + "Ġdragon": 1775, + "Ġthinks": 1776, + "Ġasks": 1777, + "Ġacross": 1778, + "Okay": 1779, + "Ġchair": 1780, + "Ġteac": 1781, + "Ġbott": 1782, + "Hi": 1783, + "Ġpa": 1784, + "Ġneigh": 1785, + "Ġbar": 1786, + "Ġbike": 1787, + "Ġpack": 1788, + "iting": 1789, + "Ġdisapp": 1790, + "Ġneighb": 1791, + "Ġstuck": 1792, + "de": 1793, + "Ġgif": 1794, + "Ġfruit": 1795, + "Ġbelieve": 1796, + "cy": 1797, + "Ġve": 1798, + "Ġcrying": 1799, + "ier": 1800, + "phant": 1801, + "Ġsuddenly": 1802, + "Ġsoup": 1803, + "Ġrace": 1804, + "Ġthrew": 1805, + "Sure": 1806, + "Ġfarmer": 1807, + "Ġaccident": 1808, + "Ġdirty": 1809, + "Ġelephant": 1810, + "Ġmonkey": 1811, + "Ġheal": 1812, + "weet": 1813, + "Ġrem": 1814, + "Ġpers": 1815, + "Ġsame": 1816, + "ored": 1817, + "ventually": 1818, + "ead": 1819, + "pping": 1820, + "Ġgentle": 1821, + "Ġfree": 1822, + "Ġbrown": 1823, + "Maybe": 1824, + "Ġinst": 1825, + "Ġbee": 1826, + "ix": 1827, + "Ġben": 1828, + "Ġwin": 1829, + "Ġblack": 1830, + "Ġalong": 1831, + "Ġlonger": 1832, + "ible": 1833, + "Ġspr": 1834, + "Ġclapped": 1835, + "Finally": 1836, + "Why": 1837, + "cked": 1838, + "Ġwords": 1839, + "ĠFl": 1840, + "Ġteacher": 1841, + "Ġupset": 1842, + "chool": 1843, + "Ġpiece": 1844, + "Ġwell": 1845, + "Ġcheered": 1846, + "Ġhit": 1847, + "Ġsang": 1848, + "ment": 1849, + "Ġbowl": 1850, + "Ġbite": 1851, + "Ġdoctor": 1852, + "ĠJill": 1853, + "Ġbucket": 1854, + "Mia": 1855, + "Ġbath": 1856, + "Ġcaught": 1857, + "Ġheart": 1858, + "Ġforget": 1859, + "Ġmark": 1860, + "Ġbutt": 1861, + "Ġdry": 1862, + "Ġyoung": 1863, + "Ġsea": 1864, + "Ġcour": 1865, + "Ġdropped": 1866, + "ese": 1867, + "Ġyard": 1868, + "Ġplac": 1869, + "ourn": 1870, + "Ġschool": 1871, + "Ġswings": 1872, + "bby": 1873, + "Ġwings": 1874, + "bs": 1875, + "Ġexplained": 1876, + "Ġjourn": 1877, + "Ġbring": 1878, + "Ġdrink": 1879, + "Ġstreet": 1880, + "Ġjuice": 1881, + "ung": 1882, + "Ġnearby": 1883, + "ĠEmma": 1884, + "Ġsmo": 1885, + "Ġsmart": 1886, + "Ġpushed": 1887, + "Ġstories": 1888, + "Ġdrove": 1889, + "Ġtiny": 1890, + "Ġpen": 1891, + "Ġcourse": 1892, + "Ġes": 1893, + "Ġmine": 1894, + "Ġtoday": 1895, + "Ġpocket": 1896, + "Ġyear": 1897, + "aughter": 1898, + "Ġje": 1899, + "Ġsecret": 1900, + "Ġexp": 1901, + "Ġwithout": 1902, + "Ġsinging": 1903, + "Ġwelcome": 1904, + "Ġhappen": 1905, + "to": 1906, + "Ġfit": 1907, + "Ġstu": 1908, + "vel": 1909, + "ract": 1910, + "Ġbus": 1911, + "Ġcheese": 1912, + "Ġcray": 1913, + "Ġfairy": 1914, + "Ġjar": 1915, + "Ġturns": 1916, + "Ġbush": 1917, + "bow": 1918, + "llo": 1919, + "Ġexploring": 1920, + "Ġlon": 1921, + "Ġstand": 1922, + "itting": 1923, + "Ġseemed": 1924, + "Ġspoon": 1925, + "ĠFinally": 1926, + "Ġgames": 1927, + "Ġwoke": 1928, + "issed": 1929, + "Ġresp": 1930, + "Ġtrou": 1931, + "Ġtwins": 1932, + "Ġhoped": 1933, + "Ġatt": 1934, + "Ġlonely": 1935, + "Ġant": 1936, + "ĠMama": 1937, + "orrow": 1938, + "Ġnoises": 1939, + "Ġhugs": 1940, + "ountain": 1941, + "yard": 1942, + "Ġdaughter": 1943, + "Ġinv": 1944, + "Ġkey": 1945, + "ail": 1946, + "Ġrocks": 1947, + "Ġdisco": 1948, + "Ġsu": 1949, + "Ġlunch": 1950, + "read": 1951, + "oun": 1952, + "vered": 1953, + "Ġbackyard": 1954, + "Ġsuc": 1955, + "Ġcow": 1956, + "Ġapple": 1957, + "ek": 1958, + "Ġswam": 1959, + "Ġshoes": 1960, + "Ġstars": 1961, + "Ġshook": 1962, + "ittens": 1963, + "Ġmiss": 1964, + "ches": 1965, + "Ġwish": 1966, + "Ġrelie": 1967, + "fused": 1968, + "Ġshapes": 1969, + "urple": 1970, + "Ġtower": 1971, + "ity": 1972, + "Ġcorn": 1973, + "Ġmoved": 1974, + "shine": 1975, + "Ġ3": 1976, + "Ġthough": 1977, + "Ġcir": 1978, + "umb": 1979, + "Ġtaking": 1980, + "ts": 1981, + "ĠTweet": 1982, + "ape": 1983, + "Ġrainbow": 1984, + "Ġthrow": 1985, + "Ġstar": 1986, + "Ġbench": 1987, + "Ġneck": 1988, + "ocked": 1989, + "Ġcreature": 1990, + "Ġbubble": 1991, + "Ġlate": 1992, + "Ġadventures": 1993, + "Ġowl": 1994, + "ions": 1995, + "And": 1996, + "Ġwhe": 1997, + "Ġpar": 1998, + "Ġshell": 1999, + "Ġkite": 2000, + "ĠFluffy": 2001, + "ining": 2002, + "Ġmil": 2003, + "unt": 2004, + "Ġfence": 2005, + "Ġmix": 2006, + "Ġlift": 2007, + "Ġaccidentally": 2008, + "Ġwise": 2009, + "Ġhello": 2010, + "Ġpurple": 2011, + "pper": 2012, + "Ġonto": 2013, + "Ġspin": 2014, + "ten": 2015, + "ours": 2016, + "Ġsitting": 2017, + "idge": 2018, + "Ġsunshine": 2019, + "Ġcute": 2020, + "Ġmat": 2021, + "ward": 2022, + "Ġshop": 2023, + "Ġwhist": 2024, + "Ġdist": 2025, + "Ġorange": 2026, + "Lila": 2027, + "Ġnaught": 2028, + "ĠSammy": 2029, + "Ġnaughty": 2030, + "Ġnest": 2031, + "ose": 2032, + "ors": 2033, + "lla": 2034, + "Ġmilk": 2035, + "fish": 2036, + "Ġaf": 2037, + "Ġboun": 2038, + "Ġphone": 2039, + "Ġarms": 2040, + "Ġeas": 2041, + "ness": 2042, + "Ġcarry": 2043, + "ĠGrandma": 2044, + "og": 2045, + "Ġbought": 2046, + "Ġdriver": 2047, + "Ġinstead": 2048, + "Later": 2049, + "Ġtalked": 2050, + "Ġsearched": 2051, + "gged": 2052, + "Ġpig": 2053, + "Ġcozy": 2054, + "Ġsunny": 2055, + "itty": 2056, + "Ġfeels": 2057, + "Ġfan": 2058, + "Ġwondered": 2059, + "Ġcomes": 2060, + "Ġswimming": 2061, + "uched": 2062, + "Ġtouched": 2063, + "Ġpop": 2064, + "ĠInside": 2065, + "ale": 2066, + "Ġlucky": 2067, + "Ġlaughing": 2068, + "Ġbelong": 2069, + "Ġsounds": 2070, + "Ġwom": 2071, + "Ġcave": 2072, + "let": 2073, + "Ġjourney": 2074, + "Ġwoman": 2075, + "Ġcou": 2076, + "Ġnothing": 2077, + "Ġcoat": 2078, + "Ġrelieved": 2079, + "ĠO": 2080, + "Ġpeace": 2081, + "adow": 2082, + "Ġnose": 2083, + "Ġbusy": 2084, + "Bob": 2085, + "Ġpast": 2086, + "Ġapples": 2087, + "Ġsick": 2088, + "Ġrope": 2089, + "Ġpatient": 2090, + "Ġact": 2091, + "Ġpudd": 2092, + "Ġinc": 2093, + "raid": 2094, + "Ġwatching": 2095, + "Ġbranch": 2096, + "Ġafraid": 2097, + "Ġtom": 2098, + "ider": 2099, + "Ġdays": 2100, + "Ġret": 2101, + "cing": 2102, + "ĠMary": 2103, + "Ġband": 2104, + "Ġfarm": 2105, + "Ġhuge": 2106, + "ĠThank": 2107, + "ator": 2108, + "Ġexciting": 2109, + "Ġdanced": 2110, + "thday": 2111, + "Ġwar": 2112, + "ĠSoon": 2113, + "ball": 2114, + "Ġgigg": 2115, + "blem": 2116, + "Ġzoom": 2117, + "Ġmad": 2118, + "Ġvill": 2119, + "Ġforever": 2120, + "Ġproblem": 2121, + "Ġcollect": 2122, + "ped": 2123, + "irt": 2124, + "Ġminut": 2125, + "Ġwra": 2126, + "ho": 2127, + "Ġlife": 2128, + "Ġknee": 2129, + "ces": 2130, + "Do": 2131, + "cer": 2132, + "mp": 2133, + "arp": 2134, + "Ġking": 2135, + "Ġasleep": 2136, + "Ġreturn": 2137, + "Ġsharp": 2138, + "Ġrel": 2139, + "Ġpower": 2140, + "eth": 2141, + "Ġholding": 2142, + "Ġfing": 2143, + "Ġhealthy": 2144, + "ĠMittens": 2145, + "Ġprot": 2146, + "Ġmeet": 2147, + "Ġorgan": 2148, + "Ġclever": 2149, + "Ġspotted": 2150, + "Ġletter": 2151, + "Ġgather": 2152, + "alm": 2153, + "Of": 2154, + "ere": 2155, + "Ġround": 2156, + "Ġstorm": 2157, + "Ġprotect": 2158, + "Ġgift": 2159, + "amed": 2160, + "Ġmail": 2161, + "ĠJen": 2162, + "Ġbirthday": 2163, + "Ġpres": 2164, + "Ġsaf": 2165, + "Ġneighbor": 2166, + "epend": 2167, + "Ġtakes": 2168, + "sc": 2169, + "Ġear": 2170, + "Ġcomfort": 2171, + "Ġent": 2172, + "Ġrep": 2173, + "Ġperson": 2174, + "Ġsmiling": 2175, + "ĠWith": 2176, + "Ġgently": 2177, + "Ġpointed": 2178, + "Every": 2179, + "owed": 2180, + "Ġspo": 2181, + "col": 2182, + "Ġtasty": 2183, + "ular": 2184, + "ĠJimmy": 2185, + "Ġplane": 2186, + "sting": 2187, + "hin": 2188, + "Ġhoney": 2189, + "itch": 2190, + "Ġpill": 2191, + "Ġpass": 2192, + "itten": 2193, + "Ġmatter": 2194, + "Ġadm": 2195, + "Ġtasted": 2196, + "Ġsmelled": 2197, + "nic": 2198, + "ove": 2199, + "Ġeag": 2200, + "Ġshining": 2201, + "Hey": 2202, + "Ġhope": 2203, + "Ġplaces": 2204, + "Ġkick": 2205, + "ĠCh": 2206, + "Ġpicnic": 2207, + "Ġbottle": 2208, + "Ġrespect": 2209, + "Ġblew": 2210, + "Ġappeared": 2211, + "Ġsupp": 2212, + "Tommy": 2213, + "ĠCome": 2214, + "Ġspider": 2215, + "Ġintere": 2216, + "Ġcheer": 2217, + "board": 2218, + "Ġwearing": 2219, + "Ġpresent": 2220, + "ert": 2221, + "Ġmist": 2222, + "ize": 2223, + "Ġtrouble": 2224, + "Ġtrip": 2225, + "Ġpract": 2226, + "Ġkiss": 2227, + "Ġquest": 2228, + "ond": 2229, + "Ġash": 2230, + "Ġloves": 2231, + "Ġtightly": 2232, + "gg": 2233, + "arge": 2234, + "Ġsur": 2235, + "Ġspread": 2236, + "aur": 2237, + "Ġwide": 2238, + "ones": 2239, + "umber": 2240, + "Ġfeather": 2241, + "Ġsplas": 2242, + "ont": 2243, + "ndp": 2244, + "Ġlater": 2245, + "Ġwhole": 2246, + "Ġanyone": 2247, + "Ġbread": 2248, + "Molly": 2249, + "Ġdes": 2250, + "Ġrec": 2251, + "ella": 2252, + "Ġroad": 2253, + "In": 2254, + "Ġoce": 2255, + "Ġgiant": 2256, + "Ġocean": 2257, + "els": 2258, + "Ġpuzz": 2259, + "Ġcookie": 2260, + "Ġpan": 2261, + "Ġpath": 2262, + "ÅĵI": 2263, + "Ġconfused": 2264, + "Ġscreamed": 2265, + "Ġclouds": 2266, + "eb": 2267, + "Ġimpress": 2268, + "Ġfront": 2269, + "ĠGo": 2270, + "Ġseat": 2271, + "lex": 2272, + "Ġlay": 2273, + "Ġwhich": 2274, + "la": 2275, + "Ġthin": 2276, + "Ġteach": 2277, + "erly": 2278, + "Ġdi": 2279, + "Ġanswer": 2280, + "ungle": 2281, + "Ġap": 2282, + "ssed": 2283, + "hile": 2284, + "ub": 2285, + "Ġlarge": 2286, + "Ġjungle": 2287, + "Ġpromise": 2288, + "Ġser": 2289, + "Ġber": 2290, + "Ġwild": 2291, + "Ġbark": 2292, + "Ġmach": 2293, + "oose": 2294, + "Ġbutton": 2295, + "ndpa": 2296, + "Good": 2297, + "Ġmyster": 2298, + "Ġsnake": 2299, + "Ġvillage": 2300, + "Ġhor": 2301, + "ĠEmily": 2302, + "Come": 2303, + "Ġtool": 2304, + "Ġputs": 2305, + "Ġlast": 2306, + "Ġimag": 2307, + "ĠJake": 2308, + "Lucy": 2309, + "Ġtid": 2310, + "Ġban": 2311, + "lebr": 2312, + "Ġeventually": 2313, + "Ġslid": 2314, + "Ġwave": 2315, + "Ġlive": 2316, + "Ġchased": 2317, + "ĠEverywhere": 2318, + "Ġcelebr": 2319, + "Ġcrayons": 2320, + "Just": 2321, + "Ġspent": 2322, + "Ġwrite": 2323, + "Ġgives": 2324, + "unk": 2325, + "ife": 2326, + "Ġdecor": 2327, + "Ġfoot": 2328, + "Ġdelight": 2329, + "Ġsol": 2330, + "Ġsandw": 2331, + "Ġdirt": 2332, + "ĠJenny": 2333, + "Ġtalking": 2334, + "Ġtreats": 2335, + "Ġcart": 2336, + "ching": 2337, + "Ġwaiting": 2338, + "Ġhiding": 2339, + "Ġsnowman": 2340, + "Ġinteresting": 2341, + "Ġhon": 2342, + "ony": 2343, + "Ġshelf": 2344, + "Ġtaste": 2345, + "een": 2346, + "Jim": 2347, + "Ġstep": 2348, + "Mum": 2349, + "Ġhelpful": 2350, + "Ġfeet": 2351, + "Ġshy": 2352, + "Ġgoes": 2353, + "Ġval": 2354, + "Ġemb": 2355, + "Ġstood": 2356, + "Ġshr": 2357, + "Ġbuilding": 2358, + "Ġoven": 2359, + "Ġstret": 2360, + "Ġlovely": 2361, + "Ġducks": 2362, + "Ġcorner": 2363, + "Ġwash": 2364, + "cks": 2365, + "Ġspe": 2366, + "Ġpretended": 2367, + "Ġrolled": 2368, + "Ġrude": 2369, + "old": 2370, + "Ġcalm": 2371, + "Ġpile": 2372, + "Ġteeth": 2373, + "Ġjumping": 2374, + "Ġzoo": 2375, + "Ġpuddle": 2376, + "Ġtidy": 2377, + "Mama": 2378, + "iger": 2379, + "ipe": 2380, + "Ġbugs": 2381, + "Ġashamed": 2382, + "Ġbat": 2383, + "ina": 2384, + "ĠRe": 2385, + "Ġob": 2386, + "Ġstra": 2387, + "ĠSt": 2388, + "Ġwhistle": 2389, + "Ġbanan": 2390, + "not": 2391, + "Ġnet": 2392, + "Ġforgive": 2393, + "Ġprince": 2394, + "ener": 2395, + "Ġshirt": 2396, + "Ġsel": 2397, + "Ġcoo": 2398, + "Ġembar": 2399, + "umpy": 2400, + "Ġsparkly": 2401, + "Ġremind": 2402, + "Ġsug": 2403, + "Ġcross": 2404, + "Ġmind": 2405, + "Ġcannot": 2406, + "Ġsuccess": 2407, + "Ġembarra": 2408, + "Soon": 2409, + "Ġmot": 2410, + "Ġsharing": 2411, + "Ġworm": 2412, + "Ġfold": 2413, + "Ġpeaceful": 2414, + "Ġwore": 2415, + "Ġmountain": 2416, + "Ġblow": 2417, + "lice": 2418, + "Ġign": 2419, + "Ġbell": 2420, + "Ġfurry": 2421, + "Ġgrab": 2422, + "als": 2423, + "ometimes": 2424, + "Ġwhenever": 2425, + "Ġpour": 2426, + "vous": 2427, + "llie": 2428, + "Ġpush": 2429, + "Ġbreath": 2430, + "Ġmachine": 2431, + "Ġsugar": 2432, + "Ġchick": 2433, + "Ġcreative": 2434, + "St": 2435, + "ned": 2436, + "Ġbal": 2437, + "Ġstir": 2438, + "Ġdolp": 2439, + "Ġtries": 2440, + "Now": 2441, + "Ġlad": 2442, + "ĠLe": 2443, + "Ġmed": 2444, + "Ġpool": 2445, + "Ġgener": 2446, + "Ġsal": 2447, + "ork": 2448, + "ult": 2449, + "Ġspray": 2450, + "Ġship": 2451, + "ian": 2452, + "Ġhum": 2453, + "Ġhel": 2454, + "isa": 2455, + "Ġselfish": 2456, + "Ġhar": 2457, + "Ġfaces": 2458, + "Ġmessy": 2459, + "Ġ'": 2460, + "Ġple": 2461, + "ĠSome": 2462, + "ĠHow": 2463, + "Ġgold": 2464, + "Ġchocol": 2465, + "Please": 2466, + "Ġminutes": 2467, + "Ġpillow": 2468, + "My": 2469, + "aking": 2470, + "Ġtun": 2471, + "Ġmissed": 2472, + "ighed": 2473, + "Ġvan": 2474, + "Ġfixed": 2475, + "Ġfighting": 2476, + "oin": 2477, + "Ġfig": 2478, + "Ġembarrassed": 2479, + "Ġmeant": 2480, + "Ġdrive": 2481, + "Ġtomorrow": 2482, + "Ġhours": 2483, + "Ġca": 2484, + "Ġner": 2485, + "Ġspicy": 2486, + "Ġknowing": 2487, + "ĠPlease": 2488, + "Sarah": 2489, + "Ġladder": 2490, + "ital": 2491, + "Ġhorse": 2492, + "ato": 2493, + "Ġug": 2494, + "ĠJust": 2495, + "Ġtrust": 2496, + "Ġhosp": 2497, + "Ġugly": 2498, + "Ġhospital": 2499, + "Ġchew": 2500, + "Ġbedroom": 2501, + "Ġeasy": 2502, + "Ġnervous": 2503, + "Ġrob": 2504, + "Ġthankful": 2505, + "Ġcarrots": 2506, + "Dad": 2507, + "elly": 2508, + "Ġgrew": 2509, + "Ġcolour": 2510, + "Ġbarked": 2511, + "Ġcouch": 2512, + "Ġnut": 2513, + "Ġmeadow": 2514, + "Ġtaught": 2515, + "Ġunderstood": 2516, + "Ġdisappoin": 2517, + "Ġpas": 2518, + "ano": 2519, + "Ġoften": 2520, + "Ġclass": 2521, + "Ġpassed": 2522, + "Ġknocked": 2523, + "Ġlanded": 2524, + "Ġmic": 2525, + "ird": 2526, + "Ġfear": 2527, + "Ġcoin": 2528, + "Ġmud": 2529, + "work": 2530, + "Ġdeter": 2531, + "Ġreg": 2532, + "Ġreturned": 2533, + "Ġdeterm": 2534, + "Ġhur": 2535, + "Ġfinish": 2536, + "izz": 2537, + "Ġtea": 2538, + "Ġsmooth": 2539, + "mo": 2540, + "Ġstuff": 2541, + "Ġmov": 2542, + "Ġgenerous": 2543, + "To": 2544, + "ĠOn": 2545, + "isp": 2546, + "Ġdrum": 2547, + "obby": 2548, + "Ġbubbles": 2549, + "Ġrobot": 2550, + "Ġthese": 2551, + "Ġseed": 2552, + "Ġglass": 2553, + "Ġdisappointed": 2554, + "ground": 2555, + "Ġring": 2556, + "Ġcard": 2557, + "Ġhears": 2558, + "Ġcarried": 2559, + "Ġmoon": 2560, + "Ġchocolate": 2561, + "Ġleaf": 2562, + "Ġsnacks": 2563, + "Ġcirc": 2564, + "Ġfancy": 2565, + "dle": 2566, + "Ġspend": 2567, + "ĠAnn": 2568, + "ustr": 2569, + "Ġcrab": 2570, + "Ġsongs": 2571, + "pack": 2572, + "Ġpir": 2573, + "ecially": 2574, + "osaur": 2575, + "Ġsold": 2576, + "Ġbridge": 2577, + "ĠEven": 2578, + "Ġsleepy": 2579, + "Ġhonest": 2580, + "Ġmir": 2581, + "ĠBr": 2582, + "Ġpaw": 2583, + "pecially": 2584, + "Ġav": 2585, + "ĠWhy": 2586, + "izzy": 2587, + "Ġchest": 2588, + "Ġwolf": 2589, + "Ġdinosaur": 2590, + "Ġespecially": 2591, + "Ġmap": 2592, + "Ġgas": 2593, + "ory": 2594, + "Ġchange": 2595, + "Ġgroup": 2596, + "Ġgathered": 2597, + "Ġsour": 2598, + "Ġpoor": 2599, + "Ġsplash": 2600, + "Ġwaves": 2601, + "Ġforward": 2602, + "Ġclown": 2603, + "Ġmysterious": 2604, + "ror": 2605, + "Ġbody": 2606, + "Ġfrustr": 2607, + "Ġcraw": 2608, + "Ġfake": 2609, + "Ġpin": 2610, + "Ġok": 2611, + "Ġrad": 2612, + "Ġwhisp": 2613, + "Ġtwirl": 2614, + "Ġstring": 2615, + "Ġsmelly": 2616, + "ĠTogether": 2617, + "sist": 2618, + "Ġcand": 2619, + "Ġgate": 2620, + "Ġscar": 2621, + "ĠKitty": 2622, + "Ġneckl": 2623, + "Billy": 2624, + "Ġmotor": 2625, + "ying": 2626, + "ote": 2627, + "Ġshore": 2628, + "urse": 2629, + "Ġjew": 2630, + "Ġweak": 2631, + "Ġgrumpy": 2632, + "Ġfinger": 2633, + "Ġsack": 2634, + "Ġkitten": 2635, + "Ġpolite": 2636, + "Ġglue": 2637, + "uce": 2638, + "Ġnumber": 2639, + "airs": 2640, + "Ġesc": 2641, + "Ġmirror": 2642, + "Ġsheep": 2643, + "Ġvase": 2644, + "Ġplants": 2645, + "Ġberries": 2646, + "most": 2647, + "Be": 2648, + "den": 2649, + "Ġjelly": 2650, + "ĠTheir": 2651, + "Ġemp": 2652, + "itter": 2653, + "There": 2654, + "Ġscr": 2655, + "Ġgray": 2656, + "Ġpuzzle": 2657, + "Ġnecklace": 2658, + "Ġmaybe": 2659, + "Ġplate": 2660, + "Ġalmost": 2661, + "usie": 2662, + "Ġfrustrated": 2663, + "Ġempty": 2664, + "port": 2665, + "orry": 2666, + "Ġthick": 2667, + "Ġjam": 2668, + "Ġskipped": 2669, + "ensive": 2670, + "ĠTeddy": 2671, + "aged": 2672, + "Ġcloset": 2673, + "Ġhidden": 2674, + "Ġexpensive": 2675, + "Ġdetermined": 2676, + "Ġbored": 2677, + "Ġlit": 2678, + "Ġdrew": 2679, + "Ġlegs": 2680, + "Ġhopping": 2681, + "ates": 2682, + "ĠSometimes": 2683, + "Ġsticks": 2684, + "Ġones": 2685, + "ĠAt": 2686, + "Ġbirdie": 2687, + "ĠDon": 2688, + "Ġpolice": 2689, + "Ġfruits": 2690, + "Ġtick": 2691, + "Ġaunt": 2692, + "Ġwis": 2693, + "Ġbake": 2694, + "Ġline": 2695, + "ots": 2696, + "Ġstone": 2697, + "Ġdogs": 2698, + "aul": 2699, + "Ġfigure": 2700, + "Jane": 2701, + "Ġsight": 2702, + "Ġcries": 2703, + "ĠBe": 2704, + "Ġmoving": 2705, + "Ġcontent": 2706, + "ĠBrown": 2707, + "Ġsaved": 2708, + "Ġcloth": 2709, + "?\".": 2710, + "box": 2711, + "Ġbranches": 2712, + "Ġpacked": 2713, + "Ġpowerful": 2714, + "Ġsplashed": 2715, + "Ġfeed": 2716, + "ested": 2717, + "Ġyelled": 2718, + "Where": 2719, + "dge": 2720, + "kin": 2721, + "Ġtiger": 2722, + "Ġpay": 2723, + "iddle": 2724, + "oop": 2725, + "ĠBella": 2726, + "Ġexam": 2727, + "Ġtaken": 2728, + "Ġcrane": 2729, + "Ġspoil": 2730, + "'d": 2731, + "Ġhang": 2732, + "Ġflag": 2733, + "Ġanswered": 2734, + "Ġdiscovered": 2735, + "ĠLeo": 2736, + "ush": 2737, + "Ġsave": 2738, + "Ġseek": 2739, + "Ġexcite": 2740, + "Ġpray": 2741, + "Ġjog": 2742, + "Ġstanding": 2743, + "Ġdolphin": 2744, + "Ġjack": 2745, + "Are": 2746, + "era": 2747, + "rag": 2748, + "Ġflut": 2749, + "getable": 2750, + "Ġboring": 2751, + "Ġlock": 2752, + "Ġincred": 2753, + "Ġexcitement": 2754, + "Ġsighed": 2755, + "Ġwip": 2756, + "Ġfis": 2757, + "Ġfill": 2758, + "Ġsoap": 2759, + "ĠMum": 2760, + "ola": 2761, + "Ġpleased": 2762, + "Ġjacket": 2763, + "light": 2764, + "Ġsugg": 2765, + "Ġmiddle": 2766, + "Ġeld": 2767, + "Ġwrote": 2768, + "Ġballoons": 2769, + "Ġorganized": 2770, + "que": 2771, + "Ġsailed": 2772, + "Their": 2773, + "Ġtells": 2774, + "Ġoy": 2775, + "Ġstones": 2776, + "Ġboard": 2777, + "Ġspun": 2778, + "Ġvegetable": 2779, + "lies": 2780, + "Ġind": 2781, + "iness": 2782, + "Ġworking": 2783, + "lower": 2784, + "alk": 2785, + "iff": 2786, + "Ġparrot": 2787, + "Stop": 2788, + "Ġscarf": 2789, + "Ġwagged": 2790, + "Ġclap": 2791, + "Ġmeasure": 2792, + "asing": 2793, + "Ġtears": 2794, + "body": 2795, + "Ġcomfortable": 2796, + "Ġleg": 2797, + "Ġswan": 2798, + "Ġtrue": 2799, + "Ġfright": 2800, + "Ġdrop": 2801, + "Ġsmoke": 2802, + "Ġfeathers": 2803, + "Ġincredible": 2804, + "ggs": 2805, + "ual": 2806, + "Ġtie": 2807, + "Ġcap": 2808, + "Ġlose": 2809, + "Ġpus": 2810, + "Ġeager": 2811, + "Ġmovie": 2812, + "fic": 2813, + "Ġcop": 2814, + "Ġeggs": 2815, + "oof": 2816, + "Ġrid": 2817, + "Ġcovered": 2818, + "Ġfil": 2819, + "Ġlem": 2820, + "Ġgrap": 2821, + "Ġdisappeared": 2822, + "Ġrecord": 2823, + "ĠTV": 2824, + "ĠBl": 2825, + "Ġext": 2826, + "Ġord": 2827, + "Ġgiggled": 2828, + "corn": 2829, + "Ġpri": 2830, + "Ġham": 2831, + "Ġrare": 2832, + "Ġyourself": 2833, + "Ġeng": 2834, + "Ġwhale": 2835, + "Ġknife": 2836, + "ĠLisa": 2837, + "Ġteam": 2838, + "ĠJohnny": 2839, + "zed": 2840, + "Ġtest": 2841, + "Ġbone": 2842, + "Ġdizzy": 2843, + "Ġlibr": 2844, + "Ġwrap": 2845, + "Ġquestions": 2846, + "iv": 2847, + "Ġlying": 2848, + "Ġstamp": 2849, + "Ġlicked": 2850, + "Ġnoisy": 2851, + "Ġkindness": 2852, + "Ġdiffic": 2853, + "Ġdrawing": 2854, + "apping": 2855, + "Jimmy": 2856, + "Ġstraw": 2857, + "Ġburn": 2858, + "Ġwagon": 2859, + "ately": 2860, + "Ġgrown": 2861, + "Ġrocket": 2862, + "Ġpopcorn": 2863, + "Sally": 2864, + "Ġbatter": 2865, + "Ġwand": 2866, + "Ġanc": 2867, + "ĠEnd": 2868, + "Ġsword": 2869, + "Ġhappening": 2870, + "Ġlifted": 2871, + "Ġbalance": 2872, + "Ġextra": 2873, + "Sorry": 2874, + "ha": 2875, + "Ġwander": 2876, + "Ġfat": 2877, + "ĠMark": 2878, + "ĠElla": 2879, + "Ġbecome": 2880, + "Ġtrack": 2881, + "Ġsmaller": 2882, + "Ġmistake": 2883, + "Ġtoast": 2884, + "ating": 2885, + "Ġmanaged": 2886, + "selves": 2887, + "ĠPeter": 2888, + "Ġbathroom": 2889, + "Ġharm": 2890, + "Ġdifficult": 2891, + "Ġtwist": 2892, + "Ġoffered": 2893, + "Ġruined": 2894, + "Ġmatch": 2895, + "Ġancient": 2896, + "Ġgar": 2897, + "illi": 2898, + "Ġdistance": 2899, + "Ġplayground": 2900, + "Ġlazy": 2901, + "Ġdancing": 2902, + "Ġadventur": 2903, + "af": 2904, + "ĠBa": 2905, + "Ġcalling": 2906, + "Ġcleaned": 2907, + "Ġcrown": 2908, + "ormal": 2909, + "Ġonce": 2910, + "Ġboss": 2911, + "Ġsell": 2912, + "Ġwalks": 2913, + "Ġfrightened": 2914, + "erri": 2915, + "Ġmask": 2916, + "Ġfluffy": 2917, + "emo": 2918, + "ĠZ": 2919, + "Ġfrag": 2920, + "Ġdest": 2921, + "Ġnormal": 2922, + "Ġyog": 2923, + "Ġpicks": 2924, + "Ġeye": 2925, + "Ġdrank": 2926, + "Ġcircle": 2927, + "Ġadventurous": 2928, + "Spot": 2929, + "Ġtent": 2930, + "Ġpress": 2931, + "Ġkissed": 2932, + "ict": 2933, + "Ġsciss": 2934, + "Ġgets": 2935, + "Ġtrunk": 2936, + "Ġcheck": 2937, + "Ġenjoying": 2938, + "Ġshells": 2939, + "Ġelderly": 2940, + "Ġscissors": 2941, + "Ġterri": 2942, + "Ġtoug": 2943, + "ages": 2944, + "Ġsteal": 2945, + "Me": 2946, + "ready": 2947, + "ete": 2948, + "Ġalready": 2949, + "ĠLittle": 2950, + "Ġjuicy": 2951, + "Ġstudy": 2952, + "Ġhorn": 2953, + "erp": 2954, + "Ġpizz": 2955, + "Ġallig": 2956, + "Ġlouder": 2957, + "Ġtowel": 2958, + "Ġcircles": 2959, + "Ġlead": 2960, + "Ġroar": 2961, + "Ġrose": 2962, + "Ġslipped": 2963, + "Ġtele": 2964, + "Ġfamous": 2965, + "Ġarg": 2966, + "Ġcheerful": 2967, + "ĠRex": 2968, + "Ġtough": 2969, + "ipper": 2970, + "Ġbarn": 2971, + "Ġfool": 2972, + "Ġdig": 2973, + "Ġdish": 2974, + "Ġagainst": 2975, + "Ġdeer": 2976, + "Ġsweetie": 2977, + "Ġengine": 2978, + "Ġed": 2979, + "Ġfather": 2980, + "Ġcover": 2981, + "Ġyours": 2982, + "Ġneeds": 2983, + "Joe": 2984, + "Ġbags": 2985, + "Ġspinning": 2986, + "Sue": 2987, + "Ġaw": 2988, + "anged": 2989, + "amond": 2990, + "Ġyarn": 2991, + "oph": 2992, + "Ġgiving": 2993, + "Ġdiamond": 2994, + "Ġsandwich": 2995, + "Ġindepend": 2996, + "Ġfragile": 2997, + "Who": 2998, + "cept": 2999, + "Ġnosy": 3000, + "Ġtwig": 3001, + "Ġhurts": 3002, + "Ġsuggested": 3003, + "Ġlemon": 3004, + "iest": 3005, + "ination": 3006, + "orge": 3007, + "Ġnod": 3008, + "Ġbloom": 3009, + "Ġhose": 3010, + "Ġboxes": 3011, + "Ġrealised": 3012, + "Ġfrown": 3013, + "ache": 3014, + "Ġrelax": 3015, + "Ġhammer": 3016, + "Ġadded": 3017, + "Ġweird": 3018, + "Ġmod": 3019, + "Ġcrack": 3020, + "Ġalligator": 3021, + "Amy": 3022, + "Gra": 3023, + "oot": 3024, + "Ġshake": 3025, + "rage": 3026, + "Ġlearning": 3027, + "Ġvegetables": 3028, + "Ġadd": 3029, + "Ġmelt": 3030, + "ĠSusie": 3031, + "elp": 3032, + "ĠBobby": 3033, + "Ġtrap": 3034, + "Ġcollar": 3035, + "Ġstubb": 3036, + "ique": 3037, + "Ġborrow": 3038, + "Ġpump": 3039, + "Ġloudly": 3040, + "Ġsearch": 3041, + "Ġchicken": 3042, + "Ġpizza": 3043, + "Ġindependent": 3044, + "Ġstubborn": 3045, + "Ġref": 3046, + "Ġasking": 3047, + "Ġbrilli": 3048, + "Ġdecide": 3049, + "Ġflash": 3050, + "Ġcaterp": 3051, + "Ġqueen": 3052, + "Ġsnee": 3053, + "ĠRose": 3054, + "Ġinvited": 3055, + "Ġstrawber": 3056, + "Ġfoolish": 3057, + "Ġcaterpill": 3058, + "gy": 3059, + "Ġtur": 3060, + "Ġdepend": 3061, + "Ġguit": 3062, + "Ġstri": 3063, + "Ġknight": 3064, + "Ġboys": 3065, + "Ġpee": 3066, + "Ġcomb": 3067, + "Ġclear": 3068, + "Ġunique": 3069, + "Ġbuttons": 3070, + "ĠTweety": 3071, + "Ġcolourful": 3072, + "Ġescape": 3073, + "Ġbrilliant": 3074, + "Ġrich": 3075, + "ĠBu": 3076, + "Ġsaying": 3077, + "Ġrode": 3078, + "Ġskin": 3079, + "Ġgarage": 3080, + "Help": 3081, + "dom": 3082, + "Ġmummy": 3083, + "Ġbeak": 3084, + "ured": 3085, + "aw": 3086, + "eorge": 3087, + "rop": 3088, + "edient": 3089, + "Ġunus": 3090, + "Ġoyster": 3091, + "Ġturkey": 3092, + "ĠOK": 3093, + "Ġfountain": 3094, + "Ġpatter": 3095, + "Ġweek": 3096, + "Ġchase": 3097, + "Ġrubb": 3098, + "Ġcounted": 3099, + "Ġregular": 3100, + "Ġwiped": 3101, + "Ġguitar": 3102, + "Ġunusual": 3103, + "land": 3104, + "Ġumb": 3105, + "Ġfork": 3106, + "Ġshows": 3107, + "Ġlights": 3108, + "Ġwalls": 3109, + "Ġdreamed": 3110, + "Ġbanana": 3111, + "Ġsuccessful": 3112, + "Ġjewel": 3113, + "Ġpattern": 3114, + "Ġumbre": 3115, + "cil": 3116, + "ray": 3117, + "inal": 3118, + "Ġlow": 3119, + "imed": 3120, + "Ġswe": 3121, + "Ġcryst": 3122, + "Ġpencil": 3123, + "Ġsafely": 3124, + "Ġtools": 3125, + "Well": 3126, + "Ġtape": 3127, + "Ġstream": 3128, + "ceed": 3129, + "Ġresc": 3130, + "Ġabove": 3131, + "Ġharder": 3132, + "Ġguess": 3133, + "Ġbottom": 3134, + "Ġmarket": 3135, + "Ġimpressed": 3136, + "Ġwaff": 3137, + "anut": 3138, + "iginal": 3139, + "Ġnotice": 3140, + "illa": 3141, + "Ġnods": 3142, + "Ġpeanut": 3143, + "olog": 3144, + "Ġpatch": 3145, + "Ġcrawled": 3146, + "Ġcaterpillar": 3147, + "na": 3148, + "Ġtank": 3149, + "Ġtask": 3150, + "Ġtire": 3151, + "Ġgun": 3152, + "Ġload": 3153, + "Ġcompet": 3154, + "umbled": 3155, + "Ġapolog": 3156, + "Ġsolve": 3157, + "How": 3158, + "Mummy": 3159, + "fast": 3160, + "hip": 3161, + "ĠNo": 3162, + "Ġbitter": 3163, + "arl": 3164, + "Ġears": 3165, + "Ġshone": 3166, + "Ġbrush": 3167, + "ĠFin": 3168, + "Ġoriginal": 3169, + "Ġfier": 3170, + "Ġglow": 3171, + "ĠNemo": 3172, + "Ġbreakfast": 3173, + "Ġadmired": 3174, + "ĠBlue": 3175, + "Ġedge": 3176, + "Ġdependable": 3177, + "und": 3178, + "uth": 3179, + "ĠIf": 3180, + "Ġclay": 3181, + "Ġstronger": 3182, + "Ġbossy": 3183, + "Ġcudd": 3184, + "aser": 3185, + "Ġgrandpa": 3186, + "Ġknows": 3187, + "Ġpicking": 3188, + "Ġjoke": 3189, + "Ġboats": 3190, + "Ġbald": 3191, + "Ġmedic": 3192, + "Ġscrat": 3193, + "Ġcone": 3194, + "Ġpenny": 3195, + "Ġlean": 3196, + "Ġraft": 3197, + "uggled": 3198, + "Ġsucceed": 3199, + "Ġobedient": 3200, + "Ġhumble": 3201, + "Ġlibrary": 3202, + "Re": 3203, + "head": 3204, + "Ġph": 3205, + "Ġnumb": 3206, + "Ġsauce": 3207, + "ĠBear": 3208, + "Ġtray": 3209, + "Ġproudly": 3210, + "Here": 3211, + "Ġsounded": 3212, + "Ġsquare": 3213, + "Ġterrible": 3214, + "Ġcrystal": 3215, + "Ġfierce": 3216, + "Jill": 3217, + "Ġwitch": 3218, + "inary": 3219, + "Ġhappiness": 3220, + "ĠLola": 3221, + "Ġhanded": 3222, + "Ġfridge": 3223, + "fa": 3224, + "Ġrough": 3225, + "ention": 3226, + "Ġchanged": 3227, + "Ġjoined": 3228, + "Ġraven": 3229, + "Ġmighty": 3230, + "Ġchose": 3231, + "Ġwheel": 3232, + "Ġumbrella": 3233, + "Ġtag": 3234, + "Ġcri": 3235, + "Ġow": 3236, + "arry": 3237, + "imp": 3238, + "ĠBill": 3239, + "Ġhelps": 3240, + "Ġfinding": 3241, + "Ġcleaning": 3242, + "Ġsquee": 3243, + "Ġfirework": 3244, + "Ġmedicine": 3245, + "azz": 3246, + "met": 3247, + "Ġplayful": 3248, + "Ġlocked": 3249, + "Ġgoat": 3250, + "Ġaccept": 3251, + "Ġcompass": 3252, + "Ġsurf": 3253, + "Ġcompetit": 3254, + "ext": 3255, + "kn": 3256, + "Ġsy": 3257, + "Ġsil": 3258, + "Ġbeet": 3259, + "ĠMar": 3260, + "uring": 3261, + "Ġchir": 3262, + "ĠEllie": 3263, + "Ġballs": 3264, + "ÅĵLet": 3265, + "Ġbutterflies": 3266, + "Ġguard": 3267, + "Ġcrayon": 3268, + "Ġdiscover": 3269, + "Ġparade": 3270, + "Ġmixed": 3271, + "Ġvalu": 3272, + "Ġhelmet": 3273, + "ĠBrownie": 3274, + "Em": 3275, + "ject": 3276, + "squ": 3277, + "ud": 3278, + "Ġgum": 3279, + "ature": 3280, + "Ġisland": 3281, + "Ġcost": 3282, + "Ġmole": 3283, + "Ġnicely": 3284, + "Ġkinds": 3285, + "Ġraced": 3286, + "Ġstove": 3287, + "Ġlaughs": 3288, + "Ġmarch": 3289, + "Ġfalls": 3290, + "Ġpersist": 3291, + "ĠBaby": 3292, + "ophie": 3293, + "Alice": 3294, + "ience": 3295, + "ito": 3296, + "Ġcage": 3297, + "Ġgoose": 3298, + "Ġcoins": 3299, + "Ġtrump": 3300, + "Ġmosqu": 3301, + "Ġserious": 3302, + "Ġmosquito": 3303, + "Ġflex": 3304, + "Ġcro": 3305, + "elon": 3306, + "Ġexcla": 3307, + "Ġchoose": 3308, + "Ġexplored": 3309, + "Ġunl": 3310, + "Ġbrothers": 3311, + "Ġguil": 3312, + "Ġnumbers": 3313, + "Ġtrumpet": 3314, + "Ġicy": 3315, + "Ġmoms": 3316, + "iew": 3317, + "Ġview": 3318, + "Ġenorm": 3319, + "Ġfireman": 3320, + "ĠMrs": 3321, + "Ġpractice": 3322, + "Ġenormous": 3323, + "wards": 3324, + "Ġsup": 3325, + "Ġvolc": 3326, + "Ġmeans": 3327, + "Ġpointing": 3328, + "Ġvaluable": 3329, + "Ġflexible": 3330, + "Ġexclaimed": 3331, + "Ġguilty": 3332, + "yal": 3333, + "Ġcam": 3334, + "Ġpipe": 3335, + "ision": 3336, + "ĠMy": 3337, + "Ġuseful": 3338, + "Ġfavour": 3339, + "Ġcrazy": 3340, + "Ġphot": 3341, + "known": 3342, + "Ġvolcano": 3343, + "Ġwake": 3344, + "Ġcity": 3345, + "omed": 3346, + "llip": 3347, + "Ġforth": 3348, + "Ġlollip": 3349, + "Ġseal": 3350, + "Ġallowed": 3351, + "Ġscare": 3352, + "Ġbrightly": 3353, + "Ġmarble": 3354, + "Ġpopular": 3355, + "Ġflashlight": 3356, + "Ġcompassion": 3357, + "Ġlollipop": 3358, + "ĠU": 3359, + "Ġing": 3360, + "Ġhunt": 3361, + "Ġdull": 3362, + "erry": 3363, + "Ġstat": 3364, + "Ġshel": 3365, + "!\".": 3366, + "Ġunknown": 3367, + "Ġhats": 3368, + "Ġshoot": 3369, + "redient": 3370, + "Ġbelonged": 3371, + "Ġfavourite": 3372, + "Ġingredient": 3373, + "Ġpil": 3374, + "Ġloyal": 3375, + "Ġbrace": 3376, + "ĠWhenever": 3377, + "Ġdreams": 3378, + "Ġkicked": 3379, + "Ġseeds": 3380, + "Ġtwirled": 3381, + "Ġsuper": 3382, + "time": 3383, + "Ġfine": 3384, + "Ġbees": 3385, + "Ġsofa": 3386, + "Ġshark": 3387, + "ĠMike": 3388, + "Ġanx": 3389, + "Ġsadly": 3390, + "Ġshowing": 3391, + "Ġways": 3392, + "Ġtrucks": 3393, + "Ġdelicate": 3394, + "ations": 3395, + "Ġrepe": 3396, + "Ġimpressive": 3397, + "Ġpoured": 3398, + "Ġpirate": 3399, + "Ġwhispered": 3400, + "Ġharmless": 3401, + "aff": 3402, + "Ġturt": 3403, + "Ġtied": 3404, + "aroo": 3405, + "ado": 3406, + "arth": 3407, + "Ġgrace": 3408, + "Ġhandle": 3409, + "Ġenc": 3410, + "Ġdeliver": 3411, + "Benny": 3412, + "Ġrushed": 3413, + "Ġusing": 3414, + "angaroo": 3415, + "Ġshape": 3416, + "Ġbushes": 3417, + "Ġpersistent": 3418, + "Ġgor": 3419, + "ĠSp": 3420, + "Ġkangaroo": 3421, + "Ġang": 3422, + "Ġoffice": 3423, + "Ġanxious": 3424, + "Go": 3425, + "alous": 3426, + "Ġspell": 3427, + "ĠWhile": 3428, + "iable": 3429, + "Ġpotato": 3430, + "Ġjealous": 3431, + "Ġwrapped": 3432, + "Ġfour": 3433, + "Ġcurt": 3434, + "Ġlog": 3435, + "Ġcoal": 3436, + "Ġreliable": 3437, + "Ġcurtain": 3438, + "Daisy": 3439, + "Ġsudden": 3440, + "mbol": 3441, + "ÅĵYes": 3442, + "aches": 3443, + "ĠPe": 3444, + "Ġpainted": 3445, + "ĠToby": 3446, + "Ġcere": 3447, + "gu": 3448, + "Ġtummy": 3449, + "Ġthose": 3450, + "ouses": 3451, + "uddy": 3452, + "Ġcush": 3453, + "Ġmetal": 3454, + "Ġsymbol": 3455, + "Ġbeetle": 3456, + "Ġcamera": 3457, + "Ġhouses": 3458, + "ilt": 3459, + "Ġcelebrate": 3460, + "Ġdelighted": 3461, + "Ġgorilla": 3462, + "Ġtooth": 3463, + "Ġthirst": 3464, + "keep": 3465, + "Ġshiver": 3466, + "Ġreward": 3467, + "Ġspace": 3468, + "Ġtravel": 3469, + "Ġrubbed": 3470, + "Ġprint": 3471, + "Ġdriving": 3472, + "Ġmarry": 3473, + "Ġwarned": 3474, + "Ġsoldier": 3475, + "Ġmotorcy": 3476, + "cut": 3477, + "Ġson": 3478, + "Ġtor": 3479, + "arlie": 3480, + "Ġhero": 3481, + "ief": 3482, + "Ġcarp": 3483, + "Ġexcitedly": 3484, + "Ġtemp": 3485, + "ĠPete": 3486, + "ĠBobo": 3487, + "cod": 3488, + "Ġhung": 3489, + "Ġgrowing": 3490, + "upid": 3491, + "Ġthinking": 3492, + "Ġtunn": 3493, + "While": 3494, + "cu": 3495, + "ipp": 3496, + "Ġhay": 3497, + "Ġyet": 3498, + "Ġvine": 3499, + "Ġacorn": 3500, + "icycle": 3501, + "Ġprep": 3502, + "Ġreminded": 3503, + "Ġgasped": 3504, + "Ġflute": 3505, + "Ġordinary": 3506, + "Ġcrocod": 3507, + "Ġamb": 3508, + "Ġcur": 3509, + "ĠBet": 3510, + "ĠMandy": 3511, + "Ġpup": 3512, + "Ġcreatures": 3513, + "Ġhurry": 3514, + "Ġspoiled": 3515, + "Ġtimes": 3516, + "Ġmis": 3517, + "Ġrice": 3518, + "Ġstupid": 3519, + "Ġshine": 3520, + "Ġresist": 3521, + "ugged": 3522, + "Ġlaw": 3523, + "Ġskull": 3524, + "coa": 3525, + "Ġmissing": 3526, + "Ġwheat": 3527, + "Ġsupport": 3528, + "Ġingredients": 3529, + "aid": 3530, + "loo": 3531, + "Ġcase": 3532, + "ery": 3533, + "Ġwashed": 3534, + "Ġdove": 3535, + "Ġspilled": 3536, + "Ġscoot": 3537, + "Ġmeal": 3538, + "Ġmule": 3539, + "ĠDucky": 3540, + "Ġmoder": 3541, + "arsh": 3542, + "ÅĵWhat": 3543, + "Ġtrick": 3544, + "Ġsett": 3545, + "ĠTweetie": 3546, + "Ġbounce": 3547, + "Ġfilthy": 3548, + "Ġbase": 3549, + "Ġhoop": 3550, + "Ġfre": 3551, + "Ġstumbled": 3552, + "Ġsoar": 3553, + "Ġweal": 3554, + "Ġseem": 3555, + "Ġword": 3556, + "Ġlab": 3557, + "ĠDave": 3558, + "ĠFr": 3559, + "Ġbadly": 3560, + "Ġhairy": 3561, + "Ġcrawl": 3562, + "Ġlively": 3563, + "Ġsteps": 3564, + "ÅĵLetâ": 3565, + "Ġbracelet": 3566, + "Ġcereal": 3567, + "Ġthirsty": 3568, + "Ġcarpet": 3569, + "Ġwool": 3570, + "ither": 3571, + "Ġgoal": 3572, + "Ġbackpack": 3573, + "Ġtruth": 3574, + "Ġcompl": 3575, + "Ġcheap": 3576, + "Ġdisg": 3577, + "Ġplaced": 3578, + "Ġearly": 3579, + "Ġdecorate": 3580, + "Ġnuts": 3581, + "Ow": 3582, + "cream": 3583, + "Ġbump": 3584, + "Ġthemselves": 3585, + "ices": 3586, + "Ġhelpless": 3587, + "Ġclum": 3588, + "Ġscatter": 3589, + "Ġscale": 3590, + "Ġnews": 3591, + "usting": 3592, + "Ġenv": 3593, + "âĤ¬âĢ": 3594, + "Ġtelling": 3595, + "Ġswinging": 3596, + "Ġsparkled": 3597, + "Ġcarrying": 3598, + "gging": 3599, + "ines": 3600, + "ric": 3601, + "Ġfriendship": 3602, + "Ġcareless": 3603, + "Ġsne": 3604, + "Ġdoesn": 3605, + "Ġfalling": 3606, + "Ġcomput": 3607, + "Ġpale": 3608, + "Ġpeeked": 3609, + "Ġcompassionate": 3610, + "com": 3611, + "vice": 3612, + "Ġsend": 3613, + "Ġstage": 3614, + "Ġmem": 3615, + "Ġchalk": 3616, + "Ġchasing": 3617, + "Ġpebble": 3618, + "ĠEverything": 3619, + "Ġcube": 3620, + "Ġpige": 3621, + "Ġcourage": 3622, + "Ġfingers": 3623, + "Ġturtle": 3624, + "bled": 3625, + "Ġsink": 3626, + "Ġbicycle": 3627, + "Ġshaking": 3628, + "Ġsleeping": 3629, + "Ġneighbour": 3630, + "Ġmodern": 3631, + "Ġdisgusting": 3632, + "Ġclumsy": 3633, + "gest": 3634, + "Ġhook": 3635, + "Ġcher": 3636, + "Ġnail": 3637, + "Ġbiggest": 3638, + "iches": 3639, + "Ġbul": 3640, + "Ġending": 3641, + "Ġdead": 3642, + "Ġtriang": 3643, + "ulance": 3644, + "Ġspoke": 3645, + "Ġpracticed": 3646, + "Ġambulance": 3647, + "Ġcomputer": 3648, + "ibb": 3649, + "Ġlamp": 3650, + "Ġstation": 3651, + "Ġchar": 3652, + "Ġwallet": 3653, + "ĠToday": 3654, + "Ġpackage": 3655, + "Ġmicrop": 3656, + "Ġcushion": 3657, + "keeper": 3658, + "Ġcrocodile": 3659, + "Ġmicrophone": 3660, + "under": 3661, + "ĠAlex": 3662, + "Ġthoughtful": 3663, + "Ġjolly": 3664, + "Ġpengu": 3665, + "Ġpumpkin": 3666, + "Ġcostum": 3667, + "Ġfive": 3668, + "Ġdough": 3669, + "Ġpony": 3670, + "Ġol": 3671, + "imi": 3672, + "Ġstairs": 3673, + "Ġknock": 3674, + "Ġgrand": 3675, + "Ġintell": 3676, + "Ġuniver": 3677, + "Ġbrighter": 3678, + "Ġopens": 3679, + "Ġdreaming": 3680, + "Ġbounced": 3681, + "Ġquestion": 3682, + "Ġstatue": 3683, + "Ġwine": 3684, + "isk": 3685, + "ĠAlice": 3686, + "Ġcocoa": 3687, + "ĠYour": 3688, + "Ġpepper": 3689, + "Ġbeauty": 3690, + "Ġperm": 3691, + "Ġpainting": 3692, + "Ġshoe": 3693, + "Ġelev": 3694, + "Everyone": 3695, + "hino": 3696, + "Ġwealthy": 3697, + "Ġintellig": 3698, + "noon": 3699, + "Ġsuit": 3700, + "Ġmild": 3701, + "Ġnature": 3702, + "ans": 3703, + "Ġhappier": 3704, + "Ġneat": 3705, + "Ġstarts": 3706, + "ucked": 3707, + "Ġafternoon": 3708, + "Ġtorn": 3709, + "Ġelevator": 3710, + "gen": 3711, + "ti": 3712, + "Ġten": 3713, + "Ġharsh": 3714, + "Ġped": 3715, + "atient": 3716, + "orant": 3717, + "Ġrece": 3718, + "Ġjet": 3719, + "illie": 3720, + "ĠRed": 3721, + "Ġreading": 3722, + "Ġsearching": 3723, + "Ġbasketball": 3724, + "Ġbarber": 3725, + "Ġspeed": 3726, + "See": 3727, + "awn": 3728, + "Ġstack": 3729, + "ĠMummy": 3730, + "Ġclock": 3731, + "ĠGive": 3732, + "Ġmagn": 3733, + "Ġleaving": 3734, + "Ġcupboard": 3735, + "Ġdesign": 3736, + "Ġslides": 3737, + "Ġsandwiches": 3738, + "Ġcandle": 3739, + "aulif": 3740, + "Ġriding": 3741, + "Ġfresh": 3742, + "oppy": 3743, + "Ġshocked": 3744, + "Ġener": 3745, + "Ġjumps": 3746, + "Ġhaircut": 3747, + "Ġrespectful": 3748, + "Ġcooking": 3749, + "Ġticket": 3750, + "Ġencou": 3751, + "Ġtunnel": 3752, + "auliflower": 3753, + "Give": 3754, + "Ġsle": 3755, + "reen": 3756, + "Ġmay": 3757, + "Ġthief": 3758, + "Ġyawn": 3759, + "Ġfort": 3760, + "Ġalert": 3761, + "Ġsorts": 3762, + "Ġimpatient": 3763, + "Ġcreate": 3764, + "ailable": 3765, + "Ġwheels": 3766, + "Ġvalue": 3767, + "Ġprize": 3768, + "Mary": 3769, + "ye": 3770, + "het": 3771, + "Ġsent": 3772, + "Ġbent": 3773, + "ince": 3774, + "Ġcomet": 3775, + "Ġcamp": 3776, + "Ġcauliflower": 3777, + "Ġmelon": 3778, + "Ġgem": 3779, + "Ġitself": 3780, + "iron": 3781, + "Ġmush": 3782, + "Ġmonkeys": 3783, + "Ġsalad": 3784, + "Ġdestro": 3785, + "Ġrescue": 3786, + "Ġscooter": 3787, + "Ġintelligent": 3788, + "az": 3789, + "Ġdug": 3790, + "iling": 3791, + "anger": 3792, + "Ġroof": 3793, + "Ġrhino": 3794, + "Ġscold": 3795, + "aghet": 3796, + "Ġexper": 3797, + "Ġbarking": 3798, + "Ġbananas": 3799, + "Ġignorant": 3800, + "Ġmicro": 3801, + "Ġavailable": 3802, + "Ġgraceful": 3803, + "Ġmotorcycle": 3804, + "aghetti": 3805, + "hood": 3806, + "ĠSn": 3807, + "Ġearth": 3808, + "Ġboots": 3809, + "Ġspaghetti": 3810, + "ummer": 3811, + "Ġagree": 3812, + "Ġbull": 3813, + "Ġanywhere": 3814, + "ĠGeorge": 3815, + "Ġbehave": 3816, + "ĠKim": 3817, + "Ġpaid": 3818, + "scope": 3819, + "Ġstretched": 3820, + "Ġfearful": 3821, + "Ġavo": 3822, + "Ġmicroscope": 3823, + "If": 3824, + "io": 3825, + "Ġnoteb": 3826, + "ĠEventually": 3827, + "Ġolder": 3828, + "Ġsniff": 3829, + "Ġadvice": 3830, + "Ġstops": 3831, + "Ġperform": 3832, + "Ġfurther": 3833, + "Ġenvious": 3834, + "Ġpigeon": 3835, + "Ġmushroom": 3836, + "Ġnotebook": 3837, + "ubby": 3838, + "Ġpun": 3839, + "ats": 3840, + "Ġnurse": 3841, + "orable": 3842, + "Ġthunder": 3843, + "Ġclapping": 3844, + "Ġvide": 3845, + "Ġbuilt": 3846, + "ĠCl": 3847, + "Ġuncom": 3848, + "Ġshouting": 3849, + "Ġarrow": 3850, + "Ġdrawer": 3851, + "Ġprov": 3852, + "fortable": 3853, + "Ġtomato": 3854, + "Ġmeeting": 3855, + "Ġobject": 3856, + "Ġyogurt": 3857, + "Ġtemple": 3858, + "Ġuncomfortable": 3859, + "ef": 3860, + "ues": 3861, + "ĠTony": 3862, + "Ġloop": 3863, + "Ġchubby": 3864, + "Ġswitch": 3865, + "Ġskip": 3866, + "Ġadorable": 3867, + "ÅĵIt": 3868, + "Ġcounting": 3869, + "Ġcooked": 3870, + "Ġpist": 3871, + "arian": 3872, + "enry": 3873, + "Ġstared": 3874, + "Ġshut": 3875, + "icop": 3876, + "Ġquite": 3877, + "Ġsnuggled": 3878, + "ĠGrandpa": 3879, + "Ġtroubled": 3880, + "Ġgiggle": 3881, + "Ġblowing": 3882, + "Ġhelicop": 3883, + "Ġfisher": 3884, + "Ġpistol": 3885, + "Ġhelicopter": 3886, + "par": 3887, + "ĠSophie": 3888, + "ooped": 3889, + "ips": 3890, + "Ġnowhere": 3891, + "Ġforgave": 3892, + "Ġmarched": 3893, + "Ġtreasures": 3894, + "host": 3895, + "Ġpassport": 3896, + "Ġstirred": 3897, + "Ġpushing": 3898, + "Ġcompetitive": 3899, + "Ġdestroy": 3900, + "Yay": 3901, + "aked": 3902, + "oe": 3903, + "tter": 3904, + "Ġgir": 3905, + "Ġghost": 3906, + "anic": 3907, + "kelet": 3908, + "htub": 3909, + "Ġreef": 3910, + "Ġsepar": 3911, + "opard": 3912, + "play": 3913, + "Ġenvel": 3914, + "Ġbreat": 3915, + "Ġrepair": 3916, + "Ġfootball": 3917, + "Ġbathtub": 3918, + "Ġrubber": 3919, + "Ġangel": 3920, + "Ġtriangle": 3921, + "Ġled": 3922, + "Ġmill": 3923, + "ĠSh": 3924, + "chanic": 3925, + "Ġnames": 3926, + "Ġleopard": 3927, + "Ġnobody": 3928, + "Ġanyway": 3929, + "uct": 3930, + "ctus": 3931, + "Ġshouldn": 3932, + "Ġzeb": 3933, + "Ġgiven": 3934, + "Ġholds": 3935, + "Ġbarrel": 3936, + "Ġbandage": 3937, + "Ġenth": 3938, + "Daddy": 3939, + "Ġzebra": 3940, + "ives": 3941, + "ube": 3942, + "year": 3943, + "Ġpor": 3944, + "Ġpand": 3945, + "Ġotter": 3946, + "Ġribb": 3947, + "Ġshield": 3948, + "Ġmechanic": 3949, + "Ġdonâ": 3950, + "Ġdisag": 3951, + "Ġdisplay": 3952, + "Ġmusician": 3953, + "ceros": 3954, + "Ġspeak": 3955, + "Ġcab": 3956, + "Ġcactus": 3957, + "ĠBetsy": 3958, + "Ġbulb": 3959, + "Ch": 3960, + "eath": 3961, + "ikes": 3962, + "xy": 3963, + "Ġtube": 3964, + "Ġahead": 3965, + "Ġsummer": 3966, + "Ġskelet": 3967, + "Ġpurse": 3968, + "Ġknob": 3969, + "Ġdidnâ": 3970, + "Ġcouldnâ": 3971, + "Ġcolours": 3972, + "Ġwindows": 3973, + "oomy": 3974, + "Ġgifted": 3975, + "Ġcartoon": 3976, + "Ġrhinoceros": 3977, + "gl": 3978, + "heart": 3979, + "Ġsize": 3980, + "Ġbet": 3981, + "ĠMiss": 3982, + "Ġscrew": 3983, + "Ġflour": 3984, + "Ġblin": 3985, + "apa": 3986, + "Ġdishes": 3987, + "Ġraining": 3988, + "Ġhoping": 3989, + "Ġshovel": 3990, + "Ġmint": 3991, + "Ġjellyfish": 3992, + "Ġsweater": 3993, + "Ġcharming": 3994, + "Ġpenguin": 3995, + "Ġenvelope": 3996, + "Ġenthus": 3997, + "Ġcabin": 3998, + "rec": 3999, + "Ġpant": 4000, + "Ġthread": 4001, + "Ġinse": 4002, + "Ġstaff": 4003, + "Ġliz": 4004, + "Ġunpack": 4005, + "Ġwriting": 4006, + "Ġtrash": 4007, + "Ġrolling": 4008, + "Ġwaffle": 4009, + "Ġiron": 4010, + "Ġwing": 4011, + "erable": 4012, + "Ġsheet": 4013, + "ĠBuddy": 4014, + "Ġshadow": 4015, + "Ġroared": 4016, + "Ġmuff": 4017, + "Ġairport": 4018, + "Ġceiling": 4019, + "Ġattic": 4020, + "Ġorganize": 4021, + "Ġstretch": 4022, + "Ġgrapes": 4023, + "Ġobs": 4024, + "ĠAre": 4025, + "ĠWould": 4026, + "ĠInst": 4027, + "Ġpiano": 4028, + "Ġsalt": 4029, + "Ġmiserable": 4030, + "Ġcord": 4031, + "Ġdess": 4032, + "Ġrefused": 4033, + "Ġscreen": 4034, + "ĠDan": 4035, + "Ġstrugg": 4036, + "Ġneedle": 4037, + "Ġbears": 4038, + "ĠAndy": 4039, + "Ġheaded": 4040, + "Ġsticky": 4041, + "Ġfreez": 4042, + "Ġzoomed": 4043, + "Ġimagined": 4044, + "Ġmedal": 4045, + "Ġmagnet": 4046, + "reci": 4047, + "Ġobser": 4048, + "Ġer": 4049, + "Ġlives": 4050, + "ĠAl": 4051, + "ĠPat": 4052, + "Ġappreci": 4053, + "asses": 4054, + "Ġsailor": 4055, + "Ġminute": 4056, + "Ġfisherman": 4057, + "vision": 4058, + "Ġrat": 4059, + "aled": 4060, + "Ġbrus": 4061, + "Ġchance": 4062, + "Ġpost": 4063, + "Ġsungl": 4064, + "Ġcheek": 4065, + "reeze": 4066, + "Ġdeaf": 4067, + "Ġrules": 4068, + "Ġfrogs": 4069, + "Ġairpl": 4070, + "ĠGod": 4071, + "Ġstrawberry": 4072, + "Ġslept": 4073, + "Ġlizard": 4074, + "Ġsunglasses": 4075, + "Ġluck": 4076, + "Ġinf": 4077, + "Ġbeep": 4078, + "ĠBunny": 4079, + "Ġmeas": 4080, + "Ġrod": 4081, + "Ġorder": 4082, + "Ġrestless": 4083, + "Ġsandbox": 4084, + "Ġmessage": 4085, + "ĠJoey": 4086, + "Ġcomplet": 4087, + "Ġattract": 4088, + "Ġgolden": 4089, + "Grandma": 4090, + "Ġpilot": 4091, + "Ġporch": 4092, + "Little": 4093, + "ler": 4094, + "ses": 4095 + }, + "merges": [ + [ + "h", + "e" + ], + [ + "Ġ", + "t" + ], + [ + "Ġ", + "a" + ], + [ + "Ġ", + "s" + ], + [ + "Ġ", + "w" + ], + [ + "n", + "d" + ], + [ + "Ġt", + "he" + ], + [ + "e", + "d" + ], + [ + "Ġa", + "nd" + ], + [ + "Ġt", + "o" + ], + [ + "Ġ", + "b" + ], + [ + "i", + "n" + ], + [ + "Ġ", + "h" + ], + [ + "Ġw", + "a" + ], + [ + "r", + "e" + ], + [ + "Ġ", + "f" + ], + [ + "i", + "t" + ], + [ + "o", + "u" + ], + [ + "Ġ", + "c" + ], + [ + "Ġ", + "l" + ], + [ + "Ġ", + "he" + ], + [ + "Ġ", + "d" + ], + [ + "e", + "r" + ], + [ + "Ġwa", + "s" + ], + [ + "Ġ", + "m" + ], + [ + "Ġ", + "p" + ], + [ + "o", + "m" + ], + [ + "Ġ", + "T" + ], + [ + "Ġ", + "o" + ], + [ + "a", + "y" + ], + [ + "a", + "r" + ], + [ + "in", + "g" + ], + [ + "i", + "s" + ], + [ + "Ġ", + "g" + ], + [ + "i", + "l" + ], + [ + "i", + "d" + ], + [ + "a", + "t" + ], + [ + "e", + "n" + ], + [ + "Ġ", + "n" + ], + [ + "Ġs", + "a" + ], + [ + "Ġh", + "a" + ], + [ + "Ġ", + "S" + ], + [ + "i", + "m" + ], + [ + "a", + "n" + ], + [ + "ĠT", + "he" + ], + [ + "o", + "r" + ], + [ + "o", + "n" + ], + [ + "Ġ", + "it" + ], + [ + "Ġt", + "h" + ], + [ + "l", + "l" + ], + [ + "l", + "e" + ], + [ + "Ġ", + "H" + ], + [ + "Ġhe", + "r" + ], + [ + "e", + "t" + ], + [ + "o", + "t" + ], + [ + "i", + "r" + ], + [ + "ĠS", + "he" + ], + [ + "ĠH", + "e" + ], + [ + "v", + "er" + ], + [ + "e", + "s" + ], + [ + "Ġ", + "in" + ], + [ + "u", + "t" + ], + [ + "o", + "w" + ], + [ + "c", + "k" + ], + [ + "Ġ", + "e" + ], + [ + "Ġ", + "u" + ], + [ + "l", + "d" + ], + [ + "ĠThe", + "y" + ], + [ + "o", + "o" + ], + [ + "i", + "g" + ], + [ + "Ġsa", + "id" + ], + [ + "a", + "m" + ], + [ + "il", + "y" + ], + [ + "Ġb", + "e" + ], + [ + "Ġ", + "y" + ], + [ + "Ġ", + "r" + ], + [ + "Ġs", + "t" + ], + [ + "c", + "e" + ], + [ + "Ġs", + "he" + ], + [ + "Ġ", + "\"" + ], + [ + "p", + "p" + ], + [ + "k", + "e" + ], + [ + "it", + "h" + ], + [ + "O", + "n" + ], + [ + "Ġ", + "I" + ], + [ + "Ġw", + "ith" + ], + [ + "v", + "e" + ], + [ + "L", + "ily" + ], + [ + "Ġo", + "n" + ], + [ + "Ġo", + "f" + ], + [ + "Ġs", + "o" + ], + [ + "Ġh", + "is" + ], + [ + "k", + "ed" + ], + [ + "r", + "i" + ], + [ + "n", + "t" + ], + [ + "ver", + "y" + ], + [ + "Ġp", + "l" + ], + [ + "Ġd", + "ay" + ], + [ + "a", + "d" + ], + [ + "Ġy", + "ou" + ], + [ + "Ġth", + "at" + ], + [ + "Ġu", + "p" + ], + [ + "Ġha", + "d" + ], + [ + "s", + "t" + ], + [ + "Ġpl", + "ay" + ], + [ + "Ġthe", + "y" + ], + [ + "Ġ", + "Lily" + ], + [ + "Ġw", + "e" + ], + [ + "Ġm", + "om" + ], + [ + "m", + "y" + ], + [ + "Ġf", + "or" + ], + [ + "e", + "l" + ], + [ + "ou", + "ld" + ], + [ + "u", + "n" + ], + [ + "Ġ", + "B" + ], + [ + "'", + "s" + ], + [ + "it", + "t" + ], + [ + "en", + "t" + ], + [ + "Ġha", + "pp" + ], + [ + "T", + "he" + ], + [ + "c", + "h" + ], + [ + "Ġl", + "i" + ], + [ + "ou", + "t" + ], + [ + "Ġwa", + "nt" + ], + [ + "Ġs", + "h" + ], + [ + "he", + "r" + ], + [ + "l", + "y" + ], + [ + "im", + "e" + ], + [ + "itt", + "le" + ], + [ + "ou", + "nd" + ], + [ + "Ġ", + "very" + ], + [ + "Ġt", + "ime" + ], + [ + "om", + "e" + ], + [ + "Ġl", + "ittle" + ], + [ + "Ġthe", + "re" + ], + [ + "s", + "e" + ], + [ + "Ġd", + "o" + ], + [ + "Ġw", + "h" + ], + [ + "a", + "ll" + ], + [ + "Ġ", + "k" + ], + [ + "e", + "nd" + ], + [ + "a", + "l" + ], + [ + "h", + "t" + ], + [ + "Ġn", + "e" + ], + [ + "Ġ", + "re" + ], + [ + "Ġn", + "ot" + ], + [ + "Ġhapp", + "y" + ], + [ + "Ġ", + "Ċ" + ], + [ + "Ġb", + "ig" + ], + [ + "Ġ", + "M" + ], + [ + "Ġb", + "ut" + ], + [ + "Ġs", + "m" + ], + [ + "a", + "ck" + ], + [ + "Ġsa", + "w" + ], + [ + "ĠI", + "t" + ], + [ + "Ġa", + "s" + ], + [ + "Ġa", + "n" + ], + [ + "r", + "a" + ], + [ + "ri", + "end" + ], + [ + "Ġf", + "riend" + ], + [ + "id", + "e" + ], + [ + "On", + "e" + ], + [ + "r", + "y" + ], + [ + "'", + "t" + ], + [ + "v", + "ed" + ], + [ + "Ġ", + "is" + ], + [ + "On", + "ce" + ], + [ + "a", + "ke" + ], + [ + ".", + "\"" + ], + [ + "Ġwe", + "re" + ], + [ + "t", + "er" + ], + [ + "u", + "g" + ], + [ + "Ġl", + "oo" + ], + [ + "Ġl", + "o" + ], + [ + "o", + "re" + ], + [ + "e", + "c" + ], + [ + "ĠT", + "im" + ], + [ + "Ġh", + "im" + ], + [ + "Ġb", + "o" + ], + [ + "!", + "\"" + ], + [ + "Ġto", + "o" + ], + [ + "Ġg", + "o" + ], + [ + "Ġup", + "on" + ], + [ + "ir", + "l" + ], + [ + "Ġ", + "j" + ], + [ + "Ġwant", + "ed" + ], + [ + "Ġg", + "irl" + ], + [ + "Ġs", + "e" + ], + [ + "Ġ", + "out" + ], + [ + "ar", + "d" + ], + [ + "w", + "ay" + ], + [ + "Ġs", + "p" + ], + [ + "il", + "l" + ], + [ + "i", + "nd" + ], + [ + "Ġthe", + "m" + ], + [ + "Ġc", + "ould" + ], + [ + "f", + "u" + ], + [ + "he", + "n" + ], + [ + "Ġa", + "t" + ], + [ + "u", + "r" + ], + [ + "Ġd", + "id" + ], + [ + "Ġsm", + "il" + ], + [ + "Ġthe", + "ir" + ], + [ + "Ġa", + "re" + ], + [ + "Ġe", + "x" + ], + [ + "Ġ", + "A" + ], + [ + "a", + "in" + ], + [ + "Ġw", + "ent" + ], + [ + "ar", + "t" + ], + [ + "he", + "d" + ], + [ + "r", + "om" + ], + [ + "i", + "c" + ], + [ + "r", + "ound" + ], + [ + "Ġha", + "ve" + ], + [ + "Ġn", + "am" + ], + [ + "l", + "p" + ], + [ + "Ġa", + "ll" + ], + [ + "Ġ", + "J" + ], + [ + "fu", + "l" + ], + [ + "Ġk", + "n" + ], + [ + "h", + "ing" + ], + [ + "oo", + "d" + ], + [ + "Ġhe", + "lp" + ], + [ + "ig", + "ht" + ], + [ + "Ġfriend", + "s" + ], + [ + "on", + "e" + ], + [ + "ar", + "k" + ], + [ + "Ġb", + "ack" + ], + [ + "u", + "m" + ], + [ + "Ġc", + "an" + ], + [ + "Ġnam", + "ed" + ], + [ + "Ġc", + "l" + ], + [ + "?", + "\"" + ], + [ + "Ġf", + "un" + ], + [ + "a", + "re" + ], + [ + "ĠB", + "en" + ], + [ + "Ġlo", + "ved" + ], + [ + "Ġa", + "l" + ], + [ + "el", + "t" + ], + [ + "ĠTim", + "my" + ], + [ + "Ġ", + "One" + ], + [ + "o", + "p" + ], + [ + "s", + "ide" + ], + [ + "Ġl", + "e" + ], + [ + "Ġn", + "o" + ], + [ + "Ġs", + "c" + ], + [ + "ĠT", + "om" + ], + [ + "Ġf", + "elt" + ], + [ + "ou", + "g" + ], + [ + "Ġsmil", + "ed" + ], + [ + "i", + "ck" + ], + [ + "Ġas", + "ked" + ], + [ + "Y", + "ou" + ], + [ + "Ġto", + "y" + ], + [ + "Ġm", + "an" + ], + [ + "Ġa", + "round" + ], + [ + "am", + "e" + ], + [ + "Ġf", + "e" + ], + [ + "Ġs", + "ay" + ], + [ + "Ġbo", + "y" + ], + [ + "Ġs", + "ome" + ], + [ + "Ġloo", + "ked" + ], + [ + "u", + "re" + ], + [ + "om", + "et" + ], + [ + "Ġb", + "r" + ], + [ + "Ġw", + "ould" + ], + [ + "Ġm", + "e" + ], + [ + "Ġb", + "ir" + ], + [ + "Ġli", + "ke" + ], + [ + "g", + "et" + ], + [ + "Ġst", + "art" + ], + [ + "Ġr", + "o" + ], + [ + "a", + "s" + ], + [ + "Ġse", + "e" + ], + [ + "Ġ", + "W" + ], + [ + "i", + "ce" + ], + [ + "on", + "g" + ], + [ + "Ġbir", + "d" + ], + [ + "Ġs", + "omet" + ], + [ + "d", + "d" + ], + [ + "Ġw", + "or" + ], + [ + "ad", + "e" + ], + [ + "i", + "e" + ], + [ + "k", + "ing" + ], + [ + "Ġa", + "g" + ], + [ + "ow", + "n" + ], + [ + "Ġt", + "re" + ], + [ + "Ġf", + "a" + ], + [ + "Ġa", + "way" + ], + [ + "Ġwh", + "at" + ], + [ + "ing", + "s" + ], + [ + "Ġstart", + "ed" + ], + [ + "get", + "her" + ], + [ + "Ġr", + "an" + ], + [ + "Ã", + "¢" + ], + [ + "â", + "Ĥ" + ], + [ + "âĤ", + "¬" + ], + [ + "ar", + "ed" + ], + [ + "Ġm", + "ake" + ], + [ + "ĠB", + "ut" + ], + [ + "it", + "ed" + ], + [ + "i", + "f" + ], + [ + "ou", + "d" + ], + [ + "Ġm", + "ade" + ], + [ + "Ġto", + "gether" + ], + [ + "Ġsomet", + "hing" + ], + [ + "Ġex", + "c" + ], + [ + "a", + "g" + ], + [ + "Ġc", + "o" + ], + [ + "Ġp", + "ark" + ], + [ + "Ġne", + "w" + ], + [ + "Ġsa", + "d" + ], + [ + "Ġp", + "ut" + ], + [ + ",", + "\"" + ], + [ + "Ġf", + "rom" + ], + [ + "b", + "le" + ], + [ + "t", + "her" + ], + [ + "Ġp", + "r" + ], + [ + "Ġm", + "u" + ], + [ + "Ġc", + "ar" + ], + [ + "Ġh", + "ome" + ], + [ + "Ġ", + "You" + ], + [ + "Ġthe", + "n" + ], + [ + "Ġw", + "hen" + ], + [ + "Ġf", + "ound" + ], + [ + "e", + "ll" + ], + [ + "Ġo", + "ther" + ], + [ + "Ġag", + "ain" + ], + [ + "Ġc", + "h" + ], + [ + "Ġd", + "ec" + ], + [ + "Ġwh", + "o" + ], + [ + "Ġl", + "a" + ], + [ + "ri", + "ed" + ], + [ + "s", + "s" + ], + [ + "Ġg", + "ood" + ], + [ + "Ġh", + "ug" + ], + [ + "Ġ", + "L" + ], + [ + "pp", + "ed" + ], + [ + "Ġwa", + "l" + ], + [ + "e", + "p" + ], + [ + "all", + "y" + ], + [ + "Ġsay", + "s" + ], + [ + "Ġf", + "l" + ], + [ + "es", + "t" + ], + [ + "a", + "ch" + ], + [ + "Ġ", + "E" + ], + [ + "Ġexc", + "ited" + ], + [ + "p", + "l" + ], + [ + "q", + "u" + ], + [ + "oo", + "k" + ], + [ + "Ġg", + "et" + ], + [ + "oug", + "ht" + ], + [ + "Ġplay", + "ing" + ], + [ + "Ġg", + "ot" + ], + [ + "Ġs", + "w" + ], + [ + "ou", + "s" + ], + [ + "h", + "at" + ], + [ + "n", + "y" + ], + [ + "id", + "ed" + ], + [ + "u", + "ck" + ], + [ + "Ġth", + "ings" + ], + [ + "Ġe", + "very" + ], + [ + "Ġdec", + "ided" + ], + [ + "Ġc", + "ame" + ], + [ + "Ġbe", + "c" + ], + [ + "a", + "ve" + ], + [ + "r", + "o" + ], + [ + "a", + "x" + ], + [ + "Ġli", + "ked" + ], + [ + "Ġd", + "own" + ], + [ + "Ġdo", + "g" + ], + [ + "Ġsc", + "ared" + ], + [ + "Ġ", + "v" + ], + [ + "u", + "dd" + ], + [ + "u", + "st" + ], + [ + "Ġon", + "e" + ], + [ + "Ġf", + "ind" + ], + [ + "Ġb", + "l" + ], + [ + "Ġth", + "an" + ], + [ + "Ġ", + "D" + ], + [ + "ou", + "se" + ], + [ + "way", + "s" + ], + [ + "Ġkn", + "e" + ], + [ + "Ġdid", + "n" + ], + [ + "a", + "p" + ], + [ + "Ġmom", + "my" + ], + [ + "Ġc", + "are" + ], + [ + "Ġal", + "ways" + ], + [ + "Ġa", + "b" + ], + [ + "is", + "t" + ], + [ + "Ġd", + "ad" + ], + [ + "Ġfe", + "el" + ], + [ + "ar", + "a" + ], + [ + "b", + "b" + ], + [ + "ar", + "n" + ], + [ + "f", + "e" + ], + [ + "Ġyou", + "r" + ], + [ + "Ġout", + "side" + ], + [ + "u", + "e" + ], + [ + "Ġg", + "ra" + ], + [ + "an", + "t" + ], + [ + "Ġ", + "ke" + ], + [ + "ĠM", + "om" + ], + [ + "Ġtoo", + "k" + ], + [ + "Ġl", + "ot" + ], + [ + "n", + "n" + ], + [ + "Ġb", + "u" + ], + [ + "Ġab", + "out" + ], + [ + "es", + "s" + ], + [ + "Ġ", + "F" + ], + [ + "nd", + "er" + ], + [ + "Ġtre", + "e" + ], + [ + "ec", + "i" + ], + [ + "Ġloo", + "k" + ], + [ + "Ġp", + "o" + ], + [ + "Ġm", + "y" + ], + [ + "ou", + "r" + ], + [ + "Ġtoy", + "s" + ], + [ + "it", + "e" + ], + [ + "c", + "hed" + ], + [ + "Ġkne", + "w" + ], + [ + "Ġth", + "ought" + ], + [ + "en", + "ed" + ], + [ + "Ġle", + "arn" + ], + [ + "Ġin", + "t" + ], + [ + "Ġo", + "ld" + ], + [ + "Ġm", + "ore" + ], + [ + "nn", + "a" + ], + [ + "is", + "e" + ], + [ + "g", + "ed" + ], + [ + "Ġt", + "a" + ], + [ + "udd", + "en" + ], + [ + "eci", + "al" + ], + [ + "Ġsp", + "ecial" + ], + [ + "ĠM", + "ax" + ], + [ + "a", + "u" + ], + [ + "Ġw", + "ill" + ], + [ + "er", + "s" + ], + [ + "The", + "y" + ], + [ + "re", + "t" + ], + [ + "Ġp", + "e" + ], + [ + "Ġh", + "o" + ], + [ + "ĠS", + "am" + ], + [ + "Ġt", + "ake" + ], + [ + "Ġb", + "all" + ], + [ + "Ġkn", + "ow" + ], + [ + "Ġla", + "ug" + ], + [ + "f", + "ter" + ], + [ + "udden", + "ly" + ], + [ + "Ġc", + "at" + ], + [ + "Ġh", + "ow" + ], + [ + "i", + "ve" + ], + [ + "Ġt", + "r" + ], + [ + "Ġmu", + "ch" + ], + [ + "Ġan", + "y" + ], + [ + "Ġp", + "u" + ], + [ + "m", + "a" + ], + [ + "Ġs", + "l" + ], + [ + "Ġs", + "or" + ], + [ + "Ġm", + "o" + ], + [ + "v", + "en" + ], + [ + "is", + "h" + ], + [ + "Ġsh", + "ow" + ], + [ + "Ġcould", + "n" + ], + [ + "au", + "se" + ], + [ + "in", + "k" + ], + [ + "B", + "ut" + ], + [ + "Ġh", + "ouse" + ], + [ + "Ġint", + "o" + ], + [ + "um", + "p" + ], + [ + "Ġo", + "ver" + ], + [ + "Ġt", + "ried" + ], + [ + "a", + "nd" + ], + [ + "Ġe", + "at" + ], + [ + "Ġs", + "k" + ], + [ + "Ġs", + "un" + ], + [ + "Ġt", + "w" + ], + [ + "Ġcl", + "o" + ], + [ + "i", + "a" + ], + [ + "Ġr", + "un" + ], + [ + "Ġha", + "nd" + ], + [ + "ĠE", + "very" + ], + [ + "Ġ", + "en" + ], + [ + "Ġin", + "side" + ], + [ + "d", + "y" + ], + [ + "Ġ", + "if" + ], + [ + "Ġto", + "ld" + ], + [ + "Ġne", + "ver" + ], + [ + "i", + "on" + ], + [ + "Ġ", + "qu" + ], + [ + "Ġbec", + "ause" + ], + [ + "b", + "y" + ], + [ + "Ġpr", + "oud" + ], + [ + "Ġg", + "ave" + ], + [ + "Ġsor", + "ry" + ], + [ + "Ġth", + "is" + ], + [ + "Ġo", + "p" + ], + [ + "Ġplay", + "ed" + ], + [ + "at", + "e" + ], + [ + "Ġex", + "pl" + ], + [ + "Ġhe", + "ard" + ], + [ + "t", + "y" + ], + [ + "Ġo", + "r" + ], + [ + "Ġwa", + "ter" + ], + [ + "an", + "k" + ], + [ + "g", + "e" + ], + [ + "s", + "ed" + ], + [ + "Ġp", + "ick" + ], + [ + "ot", + "her" + ], + [ + "Ġro", + "om" + ], + [ + "et", + "ter" + ], + [ + "Ġj", + "ust" + ], + [ + "a", + "ce" + ], + [ + "Ġhug", + "ged" + ], + [ + "a", + "k" + ], + [ + "Ġg", + "re" + ], + [ + "he", + "re" + ], + [ + "Ġof", + "f" + ], + [ + "ĠS", + "ara" + ], + [ + "Ġp", + "ret" + ], + [ + "il", + "e" + ], + [ + "Ġe", + "ach" + ], + [ + "Ġc", + "om" + ], + [ + "Ġl", + "ong" + ], + [ + "Ġbo", + "x" + ], + [ + "or", + "t" + ], + [ + "Ġst", + "r" + ], + [ + "i", + "z" + ], + [ + "Ġu", + "nt" + ], + [ + "Ġwa", + "t" + ], + [ + "ot", + "h" + ], + [ + "Ġne", + "ed" + ], + [ + "Ġj", + "o" + ], + [ + "ĠW", + "e" + ], + [ + "T", + "om" + ], + [ + "Ġsm", + "all" + ], + [ + "in", + "e" + ], + [ + "Ġbe", + "ar" + ], + [ + "M", + "om" + ], + [ + "Ġunt", + "il" + ], + [ + "Ġn", + "ice" + ], + [ + "Ġt", + "ry" + ], + [ + "v", + "ing" + ], + [ + "u", + "c" + ], + [ + "s", + "el" + ], + [ + "oug", + "h" + ], + [ + "Ġlearn", + "ed" + ], + [ + "Ġk", + "ind" + ], + [ + "ĠA", + "nna" + ], + [ + "il", + "d" + ], + [ + "Ġf", + "o" + ], + [ + "Ġman", + "y" + ], + [ + "'", + "m" + ], + [ + "ĠJ", + "ack" + ], + [ + "Ġb", + "etter" + ], + [ + "Ġ", + "im" + ], + [ + "g", + "ry" + ], + [ + "im", + "al" + ], + [ + "Ġan", + "imal" + ], + [ + "ur", + "t" + ], + [ + "f", + "t" + ], + [ + "Ġe", + "nd" + ], + [ + "Ġs", + "n" + ], + [ + "a", + "ut" + ], + [ + "Ġt", + "e" + ], + [ + "Ġc", + "le" + ], + [ + "ĠJ", + "o" + ], + [ + "v", + "ent" + ], + [ + "ur", + "p" + ], + [ + "Ġg", + "r" + ], + [ + "Ġbe", + "aut" + ], + [ + "Ġj", + "ump" + ], + [ + "m", + "b" + ], + [ + "Ġa", + "d" + ], + [ + "re", + "am" + ], + [ + "p", + "t" + ], + [ + "ĠS", + "o" + ], + [ + "ĠH", + "er" + ], + [ + "Ġfl", + "ow" + ], + [ + "i", + "es" + ], + [ + "Ġc", + "he" + ], + [ + "Ġb", + "ra" + ], + [ + "Ġthan", + "ked" + ], + [ + "Ġe", + "ven" + ], + [ + "Ġb", + "est" + ], + [ + "Ġc", + "all" + ], + [ + "ad", + "y" + ], + [ + "H", + "e" + ], + [ + "Ġlot", + "s" + ], + [ + "Ġlaug", + "hed" + ], + [ + "sel", + "f" + ], + [ + "Ġr", + "a" + ], + [ + "Ġwa", + "y" + ], + [ + "ar", + "s" + ], + [ + "ur", + "n" + ], + [ + "Ġb", + "y" + ], + [ + "Ġfa", + "st" + ], + [ + "ll", + "y" + ], + [ + "Ġf", + "am" + ], + [ + "Ġ", + "C" + ], + [ + "v", + "es" + ], + [ + "ard", + "en" + ], + [ + "Ġg", + "arden" + ], + [ + "Ġbeaut", + "i" + ], + [ + "w", + "n" + ], + [ + "T", + "h" + ], + [ + "Ġbeauti", + "ful" + ], + [ + "Ġl", + "oud" + ], + [ + "le", + "w" + ], + [ + "Ġsk", + "y" + ], + [ + "Ġd", + "on" + ], + [ + "h", + "n" + ], + [ + "er", + "ed" + ], + [ + "in", + "y" + ], + [ + "Ġcare", + "ful" + ], + [ + "Ġlo", + "ve" + ], + [ + "Ġf", + "i" + ], + [ + "ĠThe", + "n" + ], + [ + "Å", + "ĵ" + ], + [ + "a", + "se" + ], + [ + "ec", + "t" + ], + [ + "Ġsa", + "fe" + ], + [ + "ĠA", + "nd" + ], + [ + "Ġu", + "nder" + ], + [ + "Ġc", + "ome" + ], + [ + "ĠF", + "rom" + ], + [ + "Y", + "es" + ], + [ + "ĠM", + "ia" + ], + [ + "I", + "t" + ], + [ + "m", + "e" + ], + [ + "Ġh", + "ard" + ], + [ + "Ġc", + "u" + ], + [ + "Ġw", + "o" + ], + [ + "Ġl", + "ist" + ], + [ + "Ġst", + "ay" + ], + [ + "an", + "e" + ], + [ + "s", + "h" + ], + [ + "op", + "le" + ], + [ + "Ġg", + "l" + ], + [ + "n", + "ing" + ], + [ + "Ġst", + "ill" + ], + [ + "oo", + "l" + ], + [ + "Ġh", + "urt" + ], + [ + "re", + "e" + ], + [ + "ĠH", + "is" + ], + [ + "Ġim", + "p" + ], + [ + "Ġfam", + "ily" + ], + [ + "Ġ", + "â" + ], + [ + "Ġb", + "oth" + ], + [ + "r", + "m" + ], + [ + "ig", + "h" + ], + [ + "Ġli", + "ved" + ], + [ + "he", + "s" + ], + [ + "W", + "hen" + ], + [ + "Ġpe", + "ople" + ], + [ + "Ġanimal", + "s" + ], + [ + "Ġco", + "l" + ], + [ + "Ġbra", + "ve" + ], + [ + "Ġwal", + "ked" + ], + [ + "o", + "b" + ], + [ + "T", + "im" + ], + [ + "c", + "t" + ], + [ + "Ġl", + "et" + ], + [ + "urp", + "r" + ], + [ + "ĠW", + "hen" + ], + [ + "Ġtw", + "o" + ], + [ + "Ġs", + "urpr" + ], + [ + "Ġsh", + "ould" + ], + [ + "is", + "hed" + ], + [ + "Ġb", + "ad" + ], + [ + "re", + "ss" + ], + [ + "Ġke", + "pt" + ], + [ + "Ġf", + "ore" + ], + [ + "Ġf", + "lew" + ], + [ + "Ġf", + "in" + ], + [ + "Ġst", + "or" + ], + [ + "Ġf", + "ly" + ], + [ + "a", + "st" + ], + [ + "is", + "ed" + ], + [ + "i", + "p" + ], + [ + "Ġit", + "s" + ], + [ + "l", + "ed" + ], + [ + "o", + "ck" + ], + [ + "uc", + "y" + ], + [ + "f", + "ore" + ], + [ + "Ġgo", + "ing" + ], + [ + "Ġcle", + "an" + ], + [ + "Ġd", + "an" + ], + [ + "Ġp", + "ic" + ], + [ + "Ġso", + "on" + ], + [ + "Ġcall", + "ed" + ], + [ + "Ġsh", + "are" + ], + [ + "k", + "ay" + ], + [ + "Ġan", + "gry" + ], + [ + "Ġro", + "ck" + ], + [ + "Ġc", + "on" + ], + [ + "Ġpret", + "ty" + ], + [ + "N", + "o" + ], + [ + "Ġ", + "ide" + ], + [ + "i", + "ed" + ], + [ + "il", + "ly" + ], + [ + "Ġg", + "round" + ], + [ + "x", + "t" + ], + [ + "Ġr", + "ed" + ], + [ + "Ġexpl", + "ore" + ], + [ + "Ġc", + "ry" + ], + [ + "Ġad", + "vent" + ], + [ + "Ġst", + "o" + ], + [ + "s", + "o" + ], + [ + "Ġre", + "al" + ], + [ + "L", + "et" + ], + [ + "Ġw", + "ind" + ], + [ + "Ġsh", + "iny" + ], + [ + "b", + "e" + ], + [ + "Ġb", + "ook" + ], + [ + "Ġal", + "so" + ], + [ + "Ġdo", + "ll" + ], + [ + "Ġide", + "a" + ], + [ + "Ġbe", + "fore" + ], + [ + "Ġop", + "ened" + ], + [ + "dd", + "ed" + ], + [ + "Ġwh", + "ile" + ], + [ + "um", + "my" + ], + [ + "Ġke", + "ep" + ], + [ + "Ġe", + "y" + ], + [ + "Ġn", + "ow" + ], + [ + "Ġdo", + "or" + ], + [ + "Ġfeel", + "ing" + ], + [ + "âĤ¬", + "â" + ], + [ + "o", + "on" + ], + [ + "o", + "y" + ], + [ + "Ġwal", + "king" + ], + [ + "Ġno", + "ise" + ], + [ + "Ġf", + "r" + ], + [ + "le", + "s" + ], + [ + "ag", + "e" + ], + [ + "i", + "ous" + ], + [ + "Ġcol", + "or" + ], + [ + "Ġt", + "urn" + ], + [ + "t", + "hing" + ], + [ + "f", + "f" + ], + [ + "u", + "ch" + ], + [ + "t", + "h" + ], + [ + "Ġb", + "ed" + ], + [ + "ar", + "y" + ], + [ + "Ġd", + "ra" + ], + [ + "Ġpick", + "ed" + ], + [ + "im", + "b" + ], + [ + "e", + "et" + ], + [ + "Ġcl", + "imb" + ], + [ + "Ġd", + "el" + ], + [ + "W", + "hat" + ], + [ + "Ġbe", + "ing" + ], + [ + "Ġf", + "ood" + ], + [ + "Ġu", + "n" + ], + [ + "Ġf", + "ar" + ], + [ + "t", + "ure" + ], + [ + "j", + "oy" + ], + [ + "Ġadvent", + "ure" + ], + [ + "a", + "c" + ], + [ + "Ġsmil", + "e" + ], + [ + "Ġd", + "if" + ], + [ + "Ħ", + "¢" + ], + [ + "âĤ¬â", + "Ħ¢" + ], + [ + "me", + "mb" + ], + [ + "Ġw", + "r" + ], + [ + "Ġth", + "r" + ], + [ + "ug", + "ht" + ], + [ + "Ġloo", + "king" + ], + [ + "Ġne", + "xt" + ], + [ + "ic", + "ed" + ], + [ + "ĠL", + "ucy" + ], + [ + "Ġno", + "dded" + ], + [ + "Ġqu", + "ick" + ], + [ + "Ġ", + "P" + ], + [ + "Ġd", + "is" + ], + [ + "Ġre", + "pl" + ], + [ + "ĠD", + "ad" + ], + [ + "Ġwa", + "it" + ], + [ + "iz", + "ed" + ], + [ + "Ġfore", + "st" + ], + [ + "Ġclo", + "s" + ], + [ + "ĠS", + "uddenly" + ], + [ + "Ġt", + "ra" + ], + [ + "T", + "hat" + ], + [ + "Ġey", + "es" + ], + [ + "g", + "er" + ], + [ + "bb", + "it" + ], + [ + "t", + "ed" + ], + [ + "Ġo", + "wn" + ], + [ + "Ġr", + "ain" + ], + [ + "Ġimp", + "ort" + ], + [ + "Ġgre", + "at" + ], + [ + "Ġre", + "memb" + ], + [ + "Ġpic", + "ture" + ], + [ + "Th", + "ank" + ], + [ + "S", + "uddenly" + ], + [ + "Ġsto", + "pped" + ], + [ + "Ġen", + "joy" + ], + [ + "Ġv", + "o" + ], + [ + "B", + "en" + ], + [ + "Ġg", + "ive" + ], + [ + "Ġimport", + "ant" + ], + [ + "Ġwor", + "k" + ], + [ + "Ġne", + "ar" + ], + [ + "g", + "an" + ], + [ + "p", + "ot" + ], + [ + "Ġe", + "ver" + ], + [ + "Ġa", + "pp" + ], + [ + "Ġa", + "fter" + ], + [ + "Ġquick", + "ly" + ], + [ + "Ġlist", + "en" + ], + [ + "Ġb", + "re" + ], + [ + "t", + "ing" + ], + [ + "bb", + "ed" + ], + [ + "Ġm", + "a" + ], + [ + "Ġf", + "ish" + ], + [ + "Ġrepl", + "ied" + ], + [ + "Ġra", + "bbit" + ], + [ + "Ġhand", + "s" + ], + [ + "Ġ", + "G" + ], + [ + "Ġnot", + "iced" + ], + [ + "Ġbr", + "o" + ], + [ + "Ġsl", + "ide" + ], + [ + "Ġth", + "ink" + ], + [ + "Ġwal", + "k" + ], + [ + "Ġtr", + "uck" + ], + [ + "Ġa", + "c" + ], + [ + "k", + "es" + ], + [ + "Ġstr", + "ong" + ], + [ + "S", + "he" + ], + [ + "Ġshow", + "ed" + ], + [ + "Ġd", + "e" + ], + [ + "Ġevery", + "one" + ], + [ + "Ġwo", + "nder" + ], + [ + "fe", + "re" + ], + [ + "by", + "e" + ], + [ + "Ġdif", + "fere" + ], + [ + "ir", + "st" + ], + [ + "S", + "o" + ], + [ + "Ġs", + "ure" + ], + [ + "Ġha", + "s" + ], + [ + "Ġr", + "ight" + ], + [ + "Ġbe", + "en" + ], + [ + "Ġbec", + "ame" + ], + [ + "Ġs", + "ound" + ], + [ + "ma", + "z" + ], + [ + "Ġto", + "w" + ], + [ + "Ġr", + "u" + ], + [ + "Ġt", + "al" + ], + [ + "Ġa", + "maz" + ], + [ + "Ġhe", + "ad" + ], + [ + "Ġsh", + "out" + ], + [ + "Ġbr", + "ight" + ], + [ + "Ġy", + "e" + ], + [ + "Ġwat", + "ched" + ], + [ + "Ġ", + "R" + ], + [ + "Ġm", + "or" + ], + [ + "Ġch", + "ild" + ], + [ + "a", + "ble" + ], + [ + "Ġme", + "an" + ], + [ + "Ġw", + "here" + ], + [ + "ll", + "ow" + ], + [ + "Ġh", + "igh" + ], + [ + "ĠS", + "ue" + ], + [ + "Ġfa", + "ce" + ], + [ + "Ġc", + "ook" + ], + [ + "d", + "ay" + ], + [ + "ay", + "be" + ], + [ + "Ġwat", + "ch" + ], + [ + "Ġbl", + "ue" + ], + [ + "a", + "ught" + ], + [ + "Ġdiffere", + "nt" + ], + [ + "Ġst", + "ore" + ], + [ + "Ġ", + "N" + ], + [ + "Ġgood", + "bye" + ], + [ + "Ġd", + "ress" + ], + [ + "u", + "ll" + ], + [ + "ĠB", + "ob" + ], + [ + "n", + "g" + ], + [ + "an", + "ge" + ], + [ + "Ġs", + "qu" + ], + [ + "Ġo", + "kay" + ], + [ + "is", + "y" + ], + [ + "le", + "ase" + ], + [ + "ĠMom", + "my" + ], + [ + "Ġvo", + "ice" + ], + [ + "J", + "o" + ], + [ + "at", + "h" + ], + [ + "Ġn", + "ight" + ], + [ + "ĠS", + "pot" + ], + [ + "Ġu", + "s" + ], + [ + "Ġbo", + "at" + ], + [ + "Ġflow", + "ers" + ], + [ + "Ġpl", + "ace" + ], + [ + "Ġfo", + "llow" + ], + [ + "Ġa", + "r" + ], + [ + "Ġu", + "se" + ], + [ + "Ġclos", + "er" + ], + [ + "un", + "ny" + ], + [ + "le", + "ep" + ], + [ + "ir", + "ed" + ], + [ + "Ġfa", + "v" + ], + [ + "Ġy", + "ell" + ], + [ + "Ġgra", + "bbed" + ], + [ + "Ġcu", + "ri" + ], + [ + "Ġwa", + "rm" + ], + [ + "Ġc", + "r" + ], + [ + "Ġfor", + "g" + ], + [ + "ĠSara", + "h" + ], + [ + "ĠJo", + "hn" + ], + [ + "Ġm", + "ag" + ], + [ + "Ġst", + "ick" + ], + [ + "W", + "e" + ], + [ + "Ġjump", + "ed" + ], + [ + "Ġc", + "ake" + ], + [ + "m", + "ore" + ], + [ + "Ġt", + "ell" + ], + [ + "Ġany", + "more" + ], + [ + "A", + "fter" + ], + [ + "Ġbut", + "ter" + ], + [ + "nd", + "ma" + ], + [ + "Ġth", + "ree" + ], + [ + "Ġas", + "k" + ], + [ + "c", + "o" + ], + [ + "Ġ", + "our" + ], + [ + "l", + "ie" + ], + [ + "Ġcuri", + "ous" + ], + [ + "ou", + "nt" + ], + [ + "or", + "n" + ], + [ + "Ġcon", + "t" + ], + [ + "Ġfe", + "ll" + ], + [ + "a", + "ched" + ], + [ + "ĠT", + "h" + ], + [ + "Ġbird", + "s" + ], + [ + "as", + "s" + ], + [ + "H", + "er" + ], + [ + "is", + "s" + ], + [ + "Ġhelp", + "ed" + ], + [ + "ĠJ", + "ane" + ], + [ + "Ġpu", + "ll" + ], + [ + "Ġf", + "irst" + ], + [ + "it", + "c" + ], + [ + "Ġbl", + "ock" + ], + [ + "Ġh", + "op" + ], + [ + "Ġb", + "it" + ], + [ + "Ġd", + "r" + ], + [ + "Ġreal", + "ized" + ], + [ + "Ġk", + "id" + ], + [ + "L", + "ook" + ], + [ + "il", + "a" + ], + [ + "Ġm", + "on" + ], + [ + "Ġbr", + "other" + ], + [ + "A", + "nna" + ], + [ + "S", + "ara" + ], + [ + "Ġ", + "z" + ], + [ + "ĠEvery", + "one" + ], + [ + "Ġat", + "e" + ], + [ + "Ġdo", + "es" + ], + [ + "im", + "es" + ], + [ + "Ġhapp", + "ened" + ], + [ + "Ġst", + "op" + ], + [ + "z", + "y" + ], + [ + "Ġy", + "ummy" + ], + [ + "Ġfav", + "or" + ], + [ + "pp", + "y" + ], + [ + "Ġk", + "itc" + ], + [ + "Ġkitc", + "hen" + ], + [ + "Ġsw", + "eet" + ], + [ + "u", + "s" + ], + [ + "Ġp", + "er" + ], + [ + "Ġre", + "ally" + ], + [ + "a", + "isy" + ], + [ + "Ġgra", + "ss" + ], + [ + "Ġfavor", + "ite" + ], + [ + "Ġbe", + "gan" + ], + [ + "Ġre", + "st" + ], + [ + "Ġre", + "ady" + ], + [ + "Ġle", + "a" + ], + [ + "Ġre", + "ached" + ], + [ + "Ġunder", + "st" + ], + [ + "a", + "ir" + ], + [ + "Ġst", + "e" + ], + [ + "Ġb", + "unny" + ], + [ + "ĠA", + "s" + ], + [ + "Ġstor", + "y" + ], + [ + "'", + "re" + ], + [ + "Ġp", + "ain" + ], + [ + "Ġs", + "ing" + ], + [ + "st", + "er" + ], + [ + "Ġhe", + "re" + ], + [ + "Ġsa", + "nd" + ], + [ + "Ġon", + "ly" + ], + [ + "Ġfl", + "o" + ], + [ + "Ġa", + "m" + ], + [ + "Ġgl", + "ad" + ], + [ + "Ġt", + "ri" + ], + [ + "Ġbe", + "h" + ], + [ + "Ġwor", + "ld" + ], + [ + "Ġop", + "en" + ], + [ + "Ġc", + "re" + ], + [ + "fu", + "lly" + ], + [ + "Ġpr", + "in" + ], + [ + "w", + "here" + ], + [ + "are", + "nt" + ], + [ + "Ġflow", + "er" + ], + [ + "Ġthr", + "ough" + ], + [ + "Ġb", + "a" + ], + [ + "Ġfi", + "re" + ], + [ + "Ġd", + "one" + ], + [ + "Ġha", + "ving" + ], + [ + "Ġth", + "ing" + ], + [ + "Ġdel", + "ic" + ], + [ + "Ġhim", + "self" + ], + [ + "Ġt", + "ired" + ], + [ + "Ġp", + "arent" + ], + [ + "Ġso", + "ft" + ], + [ + "Ġf", + "ro" + ], + [ + "Tim", + "my" + ], + [ + "Ġta", + "st" + ], + [ + "Ġbutter", + "f" + ], + [ + "ĠL", + "et" + ], + [ + "Ġc", + "ut" + ], + [ + "Ġp", + "art" + ], + [ + "Ġwh", + "y" + ], + [ + "k", + "en" + ], + [ + "Ġm", + "ess" + ], + [ + "C", + "an" + ], + [ + "Ġwor", + "ry" + ], + [ + "Mom", + "my" + ], + [ + "i", + "ver" + ], + [ + "Ġd", + "in" + ], + [ + "u", + "ff" + ], + [ + "at", + "er" + ], + [ + "Ġmag", + "ic" + ], + [ + "Ġwa", + "ved" + ], + [ + "Ġshout", + "ed" + ], + [ + "Ġpo", + "nd" + ], + [ + "Ġkid", + "s" + ], + [ + "Ġh", + "at" + ], + [ + "Ġd", + "uck" + ], + [ + "Ġse", + "es" + ], + [ + "o", + "lly" + ], + [ + "ill", + "ed" + ], + [ + "Ġg", + "ame" + ], + [ + "i", + "ent" + ], + [ + "Ġma", + "king" + ], + [ + "at", + "her" + ], + [ + "Jo", + "hn" + ], + [ + "A", + "s" + ], + [ + "ak", + "es" + ], + [ + "Ġcat", + "ch" + ], + [ + "Ġse", + "en" + ], + [ + "Ġc", + "ool" + ], + [ + "at", + "ion" + ], + [ + "Ġcom", + "ing" + ], + [ + "Ġl", + "ess" + ], + [ + "Ġd", + "ark" + ], + [ + "Ġu", + "sed" + ], + [ + "ed", + "dy" + ], + [ + "Ġfi", + "x" + ], + [ + "ĠJo", + "e" + ], + [ + "Ġthan", + "k" + ], + [ + "m", + "er" + ], + [ + "Ġto", + "p" + ], + [ + "Ġl", + "ady" + ], + [ + "Ġha", + "ir" + ], + [ + "ar", + "ing" + ], + [ + "Ġp", + "rom" + ], + [ + "Ġho", + "pped" + ], + [ + "ĠC", + "an" + ], + [ + "\"", + "." + ], + [ + "ig", + "n" + ], + [ + "Ġsurpr", + "ise" + ], + [ + "Ġdra", + "w" + ], + [ + "Ġm", + "um" + ], + [ + "Ġm", + "ouse" + ], + [ + "Ġfun", + "ny" + ], + [ + "re", + "l" + ], + [ + "Ġc", + "ra" + ], + [ + "Ġ", + "-" + ], + [ + "ap", + "er" + ], + [ + "Ġf", + "ull" + ], + [ + "Ġto", + "uch" + ], + [ + "Ġl", + "ight" + ], + [ + "Ġsp", + "ot" + ], + [ + "re", + "n" + ], + [ + "Ġd", + "ro" + ], + [ + "ĠBen", + "ny" + ], + [ + "Ġparent", + "s" + ], + [ + "Ġwor", + "ked" + ], + [ + "in", + "s" + ], + [ + "one", + "y" + ], + [ + "ĠD", + "o" + ], + [ + "Ġsurpr", + "ised" + ], + [ + "Ġcare", + "fully" + ], + [ + "Ġtre", + "es" + ], + [ + "Ġfro", + "g" + ], + [ + "Ġsw", + "ing" + ], + [ + "Ġdo", + "ing" + ], + [ + "D", + "on" + ], + [ + "Ġs", + "at" + ], + [ + "in", + "ally" + ], + [ + "ĠT", + "hat" + ], + [ + "Ġ", + "ice" + ], + [ + "ĠI", + "n" + ], + [ + "Ġhe", + "ld" + ], + [ + "W", + "ow" + ], + [ + "Ġrun", + "ning" + ], + [ + "Ġpret", + "end" + ], + [ + "Ġsw", + "im" + ], + [ + "Ġs", + "et" + ], + [ + "Ġre", + "ad" + ], + [ + "Ġdelic", + "ious" + ], + [ + "Ġneed", + "ed" + ], + [ + "Ġt", + "ight" + ], + [ + "Ġsl", + "ow" + ], + [ + "Ġrememb", + "ered" + ], + [ + "Ġlo", + "st" + ], + [ + "Ġco", + "ld" + ], + [ + "Ġsm", + "ell" + ], + [ + "ard", + "s" + ], + [ + "Ġw", + "ood" + ], + [ + "Ġhapp", + "ily" + ], + [ + "h", + "y" + ], + [ + "el", + "y" + ], + [ + "Ġlook", + "s" + ], + [ + "Ġbeh", + "ind" + ], + [ + "Ġher", + "self" + ], + [ + "Ġc", + "ried" + ], + [ + "Ġenjoy", + "ed" + ], + [ + "Ġnam", + "e" + ], + [ + "as", + "k" + ], + [ + "Ġye", + "ars" + ], + [ + "Ġbu", + "y" + ], + [ + "Ġho", + "le" + ], + [ + "Ġblock", + "s" + ], + [ + "Ġd", + "ri" + ], + [ + "Ġs", + "leep" + ], + [ + "Ġg", + "i" + ], + [ + "Ġyell", + "ow" + ], + [ + "c", + "ess" + ], + [ + "Ġbutterf", + "ly" + ], + [ + "i", + "ke" + ], + [ + "u", + "ed" + ], + [ + "Ġw", + "ished" + ], + [ + "Ġper", + "f" + ], + [ + "Ġan", + "other" + ], + [ + "ĠD", + "aisy" + ], + [ + "Ġgre", + "en" + ], + [ + "Ġmo", + "ve" + ], + [ + "Ġt", + "all" + ], + [ + "Ġa", + "ir" + ], + [ + "Ġb", + "ag" + ], + [ + "Ġflo", + "or" + ], + [ + "Ġcar", + "s" + ], + [ + "Ġless", + "on" + ], + [ + "u", + "l" + ], + [ + "Ġb", + "ow" + ], + [ + "Ġar", + "ri" + ], + [ + "Ġwind", + "ow" + ], + [ + "Ġsh", + "o" + ], + [ + "Ġchild", + "ren" + ], + [ + "n", + "er" + ], + [ + "Ġfin", + "ished" + ], + [ + "Ġh", + "ill" + ], + [ + "ĠA", + "fter" + ], + [ + "and", + "y" + ], + [ + "s", + "p" + ], + [ + "en", + "s" + ], + [ + "Ġp", + "aper" + ], + [ + "Ġli", + "kes" + ], + [ + "F", + "rom" + ], + [ + "s", + "y" + ], + [ + "Ġho", + "ld" + ], + [ + "S", + "am" + ], + [ + "Ġwr", + "ong" + ], + [ + "re", + "ed" + ], + [ + "Ġcont", + "in" + ], + [ + "Ġunderst", + "and" + ], + [ + "at", + "ter" + ], + [ + "u", + "it" + ], + [ + "re", + "w" + ], + [ + "Ġg", + "ent" + ], + [ + "Ġany", + "thing" + ], + [ + "Ġclo", + "se" + ], + [ + "Ġwa", + "ll" + ], + [ + "Ġe", + "l" + ], + [ + "as", + "ure" + ], + [ + "Ġle", + "ft" + ], + [ + "Ġa", + "ble" + ], + [ + "Ġarri", + "ved" + ], + [ + "Ġh", + "un" + ], + [ + "ro", + "ss" + ], + [ + "a", + "v" + ], + [ + "Ġhe", + "ar" + ], + [ + "Ġfriend", + "ly" + ], + [ + "Ġforg", + "ot" + ], + [ + "Ġg", + "one" + ], + [ + "am", + "a" + ], + [ + "Ġcre", + "at" + ], + [ + "Ġw", + "et" + ], + [ + "Ġli", + "on" + ], + [ + "H", + "ell" + ], + [ + "Hell", + "o" + ], + [ + "Ġd", + "ream" + ], + [ + "Ġh", + "ot" + ], + [ + "b", + "o" + ], + [ + "b", + "er" + ], + [ + "ĠS", + "ally" + ], + [ + "Ġf", + "illed" + ], + [ + "Ġd", + "ir" + ], + [ + "ie", + "ld" + ], + [ + "Ġcook", + "ies" + ], + [ + "Ġlaug", + "h" + ], + [ + "Ġbro", + "ken" + ], + [ + "Ġdin", + "ner" + ], + [ + "l", + "f" + ], + [ + "Ġel", + "se" + ], + [ + "Ġh", + "id" + ], + [ + "c", + "ed" + ], + [ + "Ġp", + "ink" + ], + [ + "Ġfollow", + "ed" + ], + [ + "Ġta", + "ble" + ], + [ + "Ġm", + "ar" + ], + [ + "Ġcolor", + "s" + ], + [ + "ou", + "p" + ], + [ + "Ġmom", + "ent" + ], + [ + "Ġcontin", + "ued" + ], + [ + "Ġwonder", + "ful" + ], + [ + "ĠE", + "m" + ], + [ + "e", + "y" + ], + [ + "in", + "ed" + ], + [ + "ir", + "rel" + ], + [ + "r", + "ot" + ], + [ + "Ġfin", + "ally" + ], + [ + "J", + "ack" + ], + [ + "Ġa", + "rm" + ], + [ + "Ġm", + "ight" + ], + [ + "ĠN", + "ow" + ], + [ + "Ġc", + "ast" + ], + [ + "Ġy", + "es" + ], + [ + "Ġbu", + "ild" + ], + [ + "Ġen", + "ough" + ], + [ + "ĠB", + "illy" + ], + [ + "Ġsome", + "one" + ], + [ + "Ġperf", + "ect" + ], + [ + "Ġfa", + "ir" + ], + [ + "Ġstay", + "ed" + ], + [ + "a", + "pp" + ], + [ + "Ġmor", + "ning" + ], + [ + "ĠThe", + "re" + ], + [ + "Ġpu", + "ppy" + ], + [ + "o", + "l" + ], + [ + "Ġdad", + "dy" + ], + [ + "Ġtry", + "ing" + ], + [ + "'", + "ll" + ], + [ + "Ġj", + "u" + ], + [ + "Ġother", + "s" + ], + [ + "Ġbook", + "s" + ], + [ + "Ġag", + "reed" + ], + [ + "Ġmu", + "s" + ], + [ + "Ġsn", + "ow" + ], + [ + "Ġsqu", + "irrel" + ], + [ + "Ġc", + "ream" + ], + [ + "ro", + "om" + ], + [ + "Ġget", + "ting" + ], + [ + "Ġgra", + "ndma" + ], + [ + "ĠL", + "ila" + ], + [ + "Ġlea", + "ves" + ], + [ + "Ġba", + "by" + ], + [ + "Ġc", + "ount" + ], + [ + "ic", + "y" + ], + [ + "Ġfe", + "w" + ], + [ + "A", + "t" + ], + [ + "Ġf", + "all" + ], + [ + "The", + "n" + ], + [ + "g", + "on" + ], + [ + "Ġbre", + "ak" + ], + [ + "Ġclimb", + "ed" + ], + [ + "ri", + "es" + ], + [ + "Ġbe", + "ach" + ], + [ + "as", + "h" + ], + [ + "Ġb", + "ug" + ], + [ + "Ġp", + "lease" + ], + [ + "Ġm", + "other" + ], + [ + "Ġbr", + "ought" + ], + [ + "Ġtra", + "in" + ], + [ + "an", + "ce" + ], + [ + "ĠTh", + "is" + ], + [ + "Ġde", + "ep" + ], + [ + "Ġv", + "is" + ], + [ + "at", + "ed" + ], + [ + "s", + "hed" + ], + [ + "Ġm", + "et" + ], + [ + "Ġwas", + "n" + ], + [ + "Ġsomet", + "imes" + ], + [ + "Ġevery", + "thing" + ], + [ + "ĠTom", + "my" + ], + [ + "i", + "ch" + ], + [ + "Ġmus", + "ic" + ], + [ + "ke", + "y" + ], + [ + "l", + "ing" + ], + [ + "Th", + "is" + ], + [ + "o", + "s" + ], + [ + "Ġturn", + "ed" + ], + [ + "l", + "c" + ], + [ + "Ġr", + "ide" + ], + [ + "o", + "om" + ], + [ + "Ġs", + "ong" + ], + [ + "or", + "ing" + ], + [ + "Ġsp", + "ark" + ], + [ + "Ġf", + "ight" + ], + [ + "Ġhun", + "gry" + ], + [ + "on", + "s" + ], + [ + "Ġpicture", + "s" + ], + [ + "d", + "s" + ], + [ + "Ġm", + "akes" + ], + [ + "Ġse", + "ar" + ], + [ + "uck", + "y" + ], + [ + "Ġlist", + "ened" + ], + [ + "Ġjo", + "y" + ], + [ + "Ġpo", + "l" + ], + [ + "ĠW", + "hat" + ], + [ + "H", + "is" + ], + [ + "e", + "e" + ], + [ + "Ġcl", + "ot" + ], + [ + "Ġsa", + "il" + ], + [ + "a", + "pped" + ], + [ + "b", + "ble" + ], + [ + "Ġcom", + "p" + ], + [ + "Ġpl", + "an" + ], + [ + "Ġwant", + "s" + ], + [ + "Ġclot", + "hes" + ], + [ + "Ġpo", + "in" + ], + [ + "Ġ", + "K" + ], + [ + "z", + "z" + ], + [ + "Ġball", + "oon" + ], + [ + "Ġstr", + "ange" + ], + [ + "m", + "an" + ], + [ + "en", + "ny" + ], + [ + "Ġp", + "i" + ], + [ + "Ġgr", + "ow" + ], + [ + "Ġfly", + "ing" + ], + [ + "r", + "ow" + ], + [ + "Ġsc", + "ary" + ], + [ + "Ġtre", + "at" + ], + [ + "Ġwa", + "ited" + ], + [ + "Ġamaz", + "ed" + ], + [ + "Ġt", + "eddy" + ], + [ + "Ġcast", + "le" + ], + [ + "is", + "ter" + ], + [ + "Ġe", + "le" + ], + [ + "Ġamaz", + "ing" + ], + [ + "m", + "ed" + ], + [ + "Ġsp", + "l" + ], + [ + "O", + "h" + ], + [ + "Ġle", + "ave" + ], + [ + "Ġown", + "er" + ], + [ + "Ġprom", + "ised" + ], + [ + "Ġdan", + "ger" + ], + [ + "e", + "ver" + ], + [ + "O", + "K" + ], + [ + "Ġs", + "uch" + ], + [ + "Ġwh", + "ite" + ], + [ + "Ġp", + "at" + ], + [ + "Ġtal", + "k" + ], + [ + "uff", + "y" + ], + [ + "Ġf", + "ur" + ], + [ + "Ġp", + "ie" + ], + [ + "op", + "e" + ], + [ + "r", + "ed" + ], + [ + "Ġpull", + "ed" + ], + [ + "Ġp", + "re" + ], + [ + "Ġwood", + "s" + ], + [ + "ĠT", + "o" + ], + [ + "Ġsh", + "ared" + ], + [ + "Ġal", + "one" + ], + [ + "Ġtow", + "ards" + ], + [ + "c", + "le" + ], + [ + "ĠM", + "aybe" + ], + [ + "Ġs", + "it" + ], + [ + "Ġn", + "ap" + ], + [ + "Ġcolor", + "ful" + ], + [ + "am", + "p" + ], + [ + "h", + "one" + ], + [ + "Ġvis", + "it" + ], + [ + "ig", + "g" + ], + [ + "ug", + "g" + ], + [ + "th", + "y" + ], + [ + "en", + "ce" + ], + [ + "Ġta", + "il" + ], + [ + "ct", + "or" + ], + [ + "Ġmagic", + "al" + ], + [ + "Ġr", + "iver" + ], + [ + "Ġh", + "ig" + ], + [ + "Ġbe", + "lie" + ], + [ + "Ġgr", + "ate" + ], + [ + "u", + "ally" + ], + [ + "an", + "g" + ], + [ + "ot", + "t" + ], + [ + "Ġh", + "ide" + ], + [ + "Ġs", + "illy" + ], + [ + "Ġsmil", + "es" + ], + [ + "ow", + "er" + ], + [ + "Ġs", + "ide" + ], + [ + "M", + "ax" + ], + [ + "Ġro", + "ll" + ], + [ + "Ġg", + "u" + ], + [ + "ĠA", + "my" + ], + [ + "Ġpr", + "o" + ], + [ + "et", + "er" + ], + [ + "Ġfast", + "er" + ], + [ + "Ġgrate", + "ful" + ], + [ + "Ġtre", + "asure" + ], + [ + "as", + "ed" + ], + [ + "Ġsn", + "ack" + ], + [ + "u", + "p" + ], + [ + "Ġla", + "nd" + ], + [ + "av", + "y" + ], + [ + "Ġmor", + "al" + ], + [ + "ain", + "ed" + ], + [ + "Ġs", + "ister" + ], + [ + "ĠG", + "ra" + ], + [ + "Ġsh", + "ap" + ], + [ + "Ġpain", + "t" + ], + [ + "Ġslow", + "ly" + ], + [ + "Ġf", + "ield" + ], + [ + "Ġp", + "et" + ], + [ + "Ġs", + "ec" + ], + [ + "or", + "m" + ], + [ + "ĠJ", + "im" + ], + [ + "Ġbr", + "an" + ], + [ + "Ġm", + "in" + ], + [ + "Ġmu", + "st" + ], + [ + "Ġhelp", + "ing" + ], + [ + "Ġwor", + "ried" + ], + [ + "Ġjo", + "b" + ], + [ + "Ġfo", + "x" + ], + [ + "s", + "et" + ], + [ + "Ġm", + "oney" + ], + [ + "Ġp", + "ot" + ], + [ + "Ġqu", + "i" + ], + [ + "Ġbe", + "l" + ], + [ + "Ġac", + "c" + ], + [ + "Ġli", + "ving" + ], + [ + "Ġmon", + "ster" + ], + [ + "le", + "ss" + ], + [ + "Ġpart", + "y" + ], + [ + "e", + "ar" + ], + [ + "Ġmo", + "st" + ], + [ + "Ġwe", + "ar" + ], + [ + "Ġrememb", + "er" + ], + [ + "i", + "re" + ], + [ + "Ġs", + "ign" + ], + [ + "w", + "l" + ], + [ + "Ġcl", + "oud" + ], + [ + "Ġc", + "andy" + ], + [ + "Ġhig", + "her" + ], + [ + "i", + "pped" + ], + [ + "id", + "ent" + ], + [ + "ĠA", + "ll" + ], + [ + "d", + "er" + ], + [ + "z", + "e" + ], + [ + "Ġqui", + "et" + ], + [ + "Ġto", + "wn" + ], + [ + "un", + "ch" + ], + [ + "Ġprin", + "cess" + ], + [ + "Ġrun", + "s" + ], + [ + "ask", + "et" + ], + [ + "e", + "m" + ], + [ + "Ġevery", + "where" + ], + [ + "Ġjo", + "in" + ], + [ + "Ġste", + "pped" + ], + [ + "Ġb", + "asket" + ], + [ + "Ġl", + "ake" + ], + [ + "Ġcu", + "p" + ], + [ + "s", + "w" + ], + [ + "out", + "h" + ], + [ + "Ġcar", + "rot" + ], + [ + "Ġco", + "ll" + ], + [ + "Ġpl", + "ant" + ], + [ + "Ġsc", + "ream" + ], + [ + "ĠDad", + "dy" + ], + [ + "Ġb", + "uck" + ], + [ + "Ġhe", + "avy" + ], + [ + "Ġbig", + "ger" + ], + [ + "Ġch", + "o" + ], + [ + "Ġbl", + "ank" + ], + [ + "Ġclo", + "sed" + ], + [ + "Ġw", + "on" + ], + [ + "is", + "es" + ], + [ + "Ġan", + "sw" + ], + [ + "b", + "r" + ], + [ + "Ġblank", + "et" + ], + [ + "Ġm", + "outh" + ], + [ + "am", + "es" + ], + [ + "ec", + "es" + ], + [ + "Ġeat", + "ing" + ], + [ + "Ġpi", + "eces" + ], + [ + "ck", + "et" + ], + [ + "Ġst", + "re" + ], + [ + "Ġwe", + "lc" + ], + [ + "Ġwould", + "n" + ], + [ + "Ġre", + "ach" + ], + [ + "Ġbro", + "ke" + ], + [ + "e", + "ared" + ], + [ + "Ġdanger", + "ous" + ], + [ + "g", + "s" + ], + [ + "p", + "h" + ], + [ + "ĠM", + "olly" + ], + [ + "Ġc", + "e" + ], + [ + "ĠM", + "r" + ], + [ + "or", + "d" + ], + [ + "f", + "ort" + ], + [ + "Ġdoll", + "s" + ], + [ + "m", + "ing" + ], + [ + "Ġdan", + "ce" + ], + [ + "Ġdra", + "gon" + ], + [ + "Ġthink", + "s" + ], + [ + "Ġask", + "s" + ], + [ + "Ġac", + "ross" + ], + [ + "O", + "kay" + ], + [ + "Ġch", + "air" + ], + [ + "Ġte", + "ac" + ], + [ + "Ġb", + "ott" + ], + [ + "H", + "i" + ], + [ + "Ġp", + "a" + ], + [ + "Ġne", + "igh" + ], + [ + "Ġb", + "ar" + ], + [ + "Ġb", + "ike" + ], + [ + "Ġp", + "ack" + ], + [ + "it", + "ing" + ], + [ + "Ġdis", + "app" + ], + [ + "Ġneigh", + "b" + ], + [ + "Ġst", + "uck" + ], + [ + "d", + "e" + ], + [ + "Ġg", + "if" + ], + [ + "Ġfr", + "uit" + ], + [ + "Ġbelie", + "ve" + ], + [ + "c", + "y" + ], + [ + "Ġ", + "ve" + ], + [ + "Ġcry", + "ing" + ], + [ + "i", + "er" + ], + [ + "ph", + "ant" + ], + [ + "Ġs", + "uddenly" + ], + [ + "Ġs", + "oup" + ], + [ + "Ġr", + "ace" + ], + [ + "Ġth", + "rew" + ], + [ + "S", + "ure" + ], + [ + "Ġfar", + "mer" + ], + [ + "Ġacc", + "ident" + ], + [ + "Ġdir", + "ty" + ], + [ + "Ġele", + "phant" + ], + [ + "Ġmon", + "key" + ], + [ + "Ġhe", + "al" + ], + [ + "w", + "eet" + ], + [ + "Ġre", + "m" + ], + [ + "Ġp", + "ers" + ], + [ + "Ġsa", + "me" + ], + [ + "or", + "ed" + ], + [ + "vent", + "ually" + ], + [ + "e", + "ad" + ], + [ + "pp", + "ing" + ], + [ + "Ġgent", + "le" + ], + [ + "Ġf", + "ree" + ], + [ + "Ġbr", + "own" + ], + [ + "M", + "aybe" + ], + [ + "Ġin", + "st" + ], + [ + "Ġbe", + "e" + ], + [ + "i", + "x" + ], + [ + "Ġb", + "en" + ], + [ + "Ġw", + "in" + ], + [ + "Ġbl", + "ack" + ], + [ + "Ġal", + "ong" + ], + [ + "Ġlong", + "er" + ], + [ + "i", + "ble" + ], + [ + "Ġsp", + "r" + ], + [ + "Ġcl", + "apped" + ], + [ + "F", + "inally" + ], + [ + "W", + "hy" + ], + [ + "ck", + "ed" + ], + [ + "Ġwor", + "ds" + ], + [ + "ĠF", + "l" + ], + [ + "Ġteac", + "her" + ], + [ + "Ġup", + "set" + ], + [ + "ch", + "ool" + ], + [ + "Ġpie", + "ce" + ], + [ + "Ġwe", + "ll" + ], + [ + "Ġche", + "ered" + ], + [ + "Ġh", + "it" + ], + [ + "Ġsa", + "ng" + ], + [ + "m", + "ent" + ], + [ + "Ġbow", + "l" + ], + [ + "Ġb", + "ite" + ], + [ + "Ġdo", + "ctor" + ], + [ + "ĠJ", + "ill" + ], + [ + "Ġbuck", + "et" + ], + [ + "M", + "ia" + ], + [ + "Ġb", + "ath" + ], + [ + "Ġc", + "aught" + ], + [ + "Ġhe", + "art" + ], + [ + "Ġfor", + "get" + ], + [ + "Ġm", + "ark" + ], + [ + "Ġbut", + "t" + ], + [ + "Ġd", + "ry" + ], + [ + "Ġyou", + "ng" + ], + [ + "Ġse", + "a" + ], + [ + "Ġc", + "our" + ], + [ + "Ġdro", + "pped" + ], + [ + "es", + "e" + ], + [ + "Ġy", + "ard" + ], + [ + "Ġpl", + "ac" + ], + [ + "our", + "n" + ], + [ + "Ġs", + "chool" + ], + [ + "Ġsw", + "ings" + ], + [ + "bb", + "y" + ], + [ + "Ġw", + "ings" + ], + [ + "b", + "s" + ], + [ + "Ġexpl", + "ained" + ], + [ + "Ġj", + "ourn" + ], + [ + "Ġbr", + "ing" + ], + [ + "Ġdr", + "ink" + ], + [ + "Ġstre", + "et" + ], + [ + "Ġju", + "ice" + ], + [ + "un", + "g" + ], + [ + "Ġnear", + "by" + ], + [ + "ĠEm", + "ma" + ], + [ + "Ġsm", + "o" + ], + [ + "Ġsm", + "art" + ], + [ + "Ġpu", + "shed" + ], + [ + "Ġstor", + "ies" + ], + [ + "Ġdro", + "ve" + ], + [ + "Ġt", + "iny" + ], + [ + "Ġp", + "en" + ], + [ + "Ġcour", + "se" + ], + [ + "Ġ", + "es" + ], + [ + "Ġm", + "ine" + ], + [ + "Ġto", + "day" + ], + [ + "Ġpo", + "cket" + ], + [ + "Ġye", + "ar" + ], + [ + "aught", + "er" + ], + [ + "Ġj", + "e" + ], + [ + "Ġsec", + "ret" + ], + [ + "Ġex", + "p" + ], + [ + "Ġwith", + "out" + ], + [ + "Ġsing", + "ing" + ], + [ + "Ġwelc", + "ome" + ], + [ + "Ġhapp", + "en" + ], + [ + "t", + "o" + ], + [ + "Ġf", + "it" + ], + [ + "Ġst", + "u" + ], + [ + "ve", + "l" + ], + [ + "ra", + "ct" + ], + [ + "Ġbu", + "s" + ], + [ + "Ġche", + "ese" + ], + [ + "Ġcr", + "ay" + ], + [ + "Ġfair", + "y" + ], + [ + "Ġj", + "ar" + ], + [ + "Ġturn", + "s" + ], + [ + "Ġbu", + "sh" + ], + [ + "b", + "ow" + ], + [ + "ll", + "o" + ], + [ + "Ġexpl", + "oring" + ], + [ + "Ġl", + "on" + ], + [ + "Ġst", + "and" + ], + [ + "itt", + "ing" + ], + [ + "Ġsee", + "med" + ], + [ + "Ġsp", + "oon" + ], + [ + "ĠF", + "inally" + ], + [ + "Ġg", + "ames" + ], + [ + "Ġwo", + "ke" + ], + [ + "is", + "sed" + ], + [ + "Ġre", + "sp" + ], + [ + "Ġtr", + "ou" + ], + [ + "Ġtw", + "ins" + ], + [ + "Ġhop", + "ed" + ], + [ + "Ġat", + "t" + ], + [ + "Ġlon", + "ely" + ], + [ + "Ġa", + "nt" + ], + [ + "ĠM", + "ama" + ], + [ + "or", + "row" + ], + [ + "Ġno", + "ises" + ], + [ + "Ġhug", + "s" + ], + [ + "ount", + "ain" + ], + [ + "y", + "ard" + ], + [ + "Ġd", + "aughter" + ], + [ + "Ġin", + "v" + ], + [ + "Ġke", + "y" + ], + [ + "a", + "il" + ], + [ + "Ġrock", + "s" + ], + [ + "Ġdis", + "co" + ], + [ + "Ġs", + "u" + ], + [ + "Ġl", + "unch" + ], + [ + "re", + "ad" + ], + [ + "ou", + "n" + ], + [ + "ver", + "ed" + ], + [ + "Ġback", + "yard" + ], + [ + "Ġs", + "uc" + ], + [ + "Ġc", + "ow" + ], + [ + "Ġapp", + "le" + ], + [ + "e", + "k" + ], + [ + "Ġsw", + "am" + ], + [ + "Ġsho", + "es" + ], + [ + "Ġst", + "ars" + ], + [ + "Ġsh", + "ook" + ], + [ + "itt", + "ens" + ], + [ + "Ġm", + "iss" + ], + [ + "c", + "hes" + ], + [ + "Ġw", + "ish" + ], + [ + "Ġre", + "lie" + ], + [ + "fu", + "sed" + ], + [ + "Ġshap", + "es" + ], + [ + "urp", + "le" + ], + [ + "Ġtow", + "er" + ], + [ + "it", + "y" + ], + [ + "Ġc", + "orn" + ], + [ + "Ġmo", + "ved" + ], + [ + "sh", + "ine" + ], + [ + "Ġ", + "3" + ], + [ + "Ġth", + "ough" + ], + [ + "Ġc", + "ir" + ], + [ + "um", + "b" + ], + [ + "Ġta", + "king" + ], + [ + "t", + "s" + ], + [ + "ĠT", + "weet" + ], + [ + "ap", + "e" + ], + [ + "Ġrain", + "bow" + ], + [ + "Ġthr", + "ow" + ], + [ + "Ġst", + "ar" + ], + [ + "Ġben", + "ch" + ], + [ + "Ġne", + "ck" + ], + [ + "ock", + "ed" + ], + [ + "Ġcreat", + "ure" + ], + [ + "Ġbu", + "bble" + ], + [ + "Ġl", + "ate" + ], + [ + "Ġadventure", + "s" + ], + [ + "Ġo", + "wl" + ], + [ + "ion", + "s" + ], + [ + "A", + "nd" + ], + [ + "Ġw", + "he" + ], + [ + "Ġp", + "ar" + ], + [ + "Ġshe", + "ll" + ], + [ + "Ġk", + "ite" + ], + [ + "ĠFl", + "uffy" + ], + [ + "in", + "ing" + ], + [ + "Ġm", + "il" + ], + [ + "u", + "nt" + ], + [ + "Ġf", + "ence" + ], + [ + "Ġm", + "ix" + ], + [ + "Ġli", + "ft" + ], + [ + "Ġaccident", + "ally" + ], + [ + "Ġw", + "ise" + ], + [ + "Ġhe", + "llo" + ], + [ + "Ġp", + "urple" + ], + [ + "pp", + "er" + ], + [ + "Ġon", + "to" + ], + [ + "Ġsp", + "in" + ], + [ + "t", + "en" + ], + [ + "our", + "s" + ], + [ + "Ġs", + "itting" + ], + [ + "id", + "ge" + ], + [ + "Ġsun", + "shine" + ], + [ + "Ġcut", + "e" + ], + [ + "Ġm", + "at" + ], + [ + "w", + "ard" + ], + [ + "Ġsh", + "op" + ], + [ + "Ġwh", + "ist" + ], + [ + "Ġd", + "ist" + ], + [ + "Ġor", + "ange" + ], + [ + "L", + "ila" + ], + [ + "Ġn", + "aught" + ], + [ + "ĠSam", + "my" + ], + [ + "Ġnaught", + "y" + ], + [ + "Ġn", + "est" + ], + [ + "o", + "se" + ], + [ + "or", + "s" + ], + [ + "ll", + "a" + ], + [ + "Ġmil", + "k" + ], + [ + "f", + "ish" + ], + [ + "Ġa", + "f" + ], + [ + "Ġb", + "oun" + ], + [ + "Ġp", + "hone" + ], + [ + "Ġarm", + "s" + ], + [ + "Ġe", + "as" + ], + [ + "n", + "ess" + ], + [ + "Ġcar", + "ry" + ], + [ + "ĠGra", + "ndma" + ], + [ + "o", + "g" + ], + [ + "Ġb", + "ought" + ], + [ + "Ġdri", + "ver" + ], + [ + "Ġinst", + "ead" + ], + [ + "L", + "ater" + ], + [ + "Ġtal", + "ked" + ], + [ + "Ġsear", + "ched" + ], + [ + "g", + "ged" + ], + [ + "Ġp", + "ig" + ], + [ + "Ġco", + "zy" + ], + [ + "Ġsun", + "ny" + ], + [ + "itt", + "y" + ], + [ + "Ġfeel", + "s" + ], + [ + "Ġf", + "an" + ], + [ + "Ġwonder", + "ed" + ], + [ + "Ġcom", + "es" + ], + [ + "Ġswim", + "ming" + ], + [ + "u", + "ched" + ], + [ + "Ġto", + "uched" + ], + [ + "Ġp", + "op" + ], + [ + "ĠIn", + "side" + ], + [ + "a", + "le" + ], + [ + "Ġl", + "ucky" + ], + [ + "Ġlaug", + "hing" + ], + [ + "Ġbel", + "ong" + ], + [ + "Ġsound", + "s" + ], + [ + "Ġw", + "om" + ], + [ + "Ġc", + "ave" + ], + [ + "le", + "t" + ], + [ + "Ġjourn", + "ey" + ], + [ + "Ġwom", + "an" + ], + [ + "Ġc", + "ou" + ], + [ + "Ġnot", + "hing" + ], + [ + "Ġco", + "at" + ], + [ + "Ġrelie", + "ved" + ], + [ + "Ġ", + "O" + ], + [ + "Ġpe", + "ace" + ], + [ + "ad", + "ow" + ], + [ + "Ġno", + "se" + ], + [ + "Ġbu", + "sy" + ], + [ + "B", + "ob" + ], + [ + "Ġp", + "ast" + ], + [ + "Ġapp", + "les" + ], + [ + "Ġs", + "ick" + ], + [ + "Ġr", + "ope" + ], + [ + "Ġpat", + "ient" + ], + [ + "Ġa", + "ct" + ], + [ + "Ġp", + "udd" + ], + [ + "Ġin", + "c" + ], + [ + "ra", + "id" + ], + [ + "Ġwatch", + "ing" + ], + [ + "Ġbran", + "ch" + ], + [ + "Ġaf", + "raid" + ], + [ + "Ġto", + "m" + ], + [ + "id", + "er" + ], + [ + "Ġday", + "s" + ], + [ + "Ġre", + "t" + ], + [ + "c", + "ing" + ], + [ + "ĠM", + "ary" + ], + [ + "Ġb", + "and" + ], + [ + "Ġfar", + "m" + ], + [ + "Ġhug", + "e" + ], + [ + "ĠTh", + "ank" + ], + [ + "at", + "or" + ], + [ + "Ġexc", + "iting" + ], + [ + "Ġdan", + "ced" + ], + [ + "th", + "day" + ], + [ + "Ġwa", + "r" + ], + [ + "ĠSo", + "on" + ], + [ + "b", + "all" + ], + [ + "Ġg", + "igg" + ], + [ + "ble", + "m" + ], + [ + "Ġz", + "oom" + ], + [ + "Ġm", + "ad" + ], + [ + "Ġv", + "ill" + ], + [ + "Ġfore", + "ver" + ], + [ + "Ġpro", + "blem" + ], + [ + "Ġcoll", + "ect" + ], + [ + "p", + "ed" + ], + [ + "ir", + "t" + ], + [ + "Ġmin", + "ut" + ], + [ + "Ġw", + "ra" + ], + [ + "h", + "o" + ], + [ + "Ġli", + "fe" + ], + [ + "Ġkne", + "e" + ], + [ + "c", + "es" + ], + [ + "D", + "o" + ], + [ + "c", + "er" + ], + [ + "m", + "p" + ], + [ + "ar", + "p" + ], + [ + "Ġk", + "ing" + ], + [ + "Ġas", + "leep" + ], + [ + "Ġret", + "urn" + ], + [ + "Ġsh", + "arp" + ], + [ + "Ġre", + "l" + ], + [ + "Ġp", + "ower" + ], + [ + "et", + "h" + ], + [ + "Ġhold", + "ing" + ], + [ + "Ġf", + "ing" + ], + [ + "Ġheal", + "thy" + ], + [ + "ĠM", + "ittens" + ], + [ + "Ġpr", + "ot" + ], + [ + "Ġme", + "et" + ], + [ + "Ġor", + "gan" + ], + [ + "Ġcle", + "ver" + ], + [ + "Ġspot", + "ted" + ], + [ + "Ġl", + "etter" + ], + [ + "Ġg", + "ather" + ], + [ + "al", + "m" + ], + [ + "O", + "f" + ], + [ + "e", + "re" + ], + [ + "Ġr", + "ound" + ], + [ + "Ġstor", + "m" + ], + [ + "Ġprot", + "ect" + ], + [ + "Ġgif", + "t" + ], + [ + "am", + "ed" + ], + [ + "Ġma", + "il" + ], + [ + "ĠJ", + "en" + ], + [ + "Ġbir", + "thday" + ], + [ + "Ġpre", + "s" + ], + [ + "Ġsa", + "f" + ], + [ + "Ġneighb", + "or" + ], + [ + "ep", + "end" + ], + [ + "Ġta", + "kes" + ], + [ + "s", + "c" + ], + [ + "Ġe", + "ar" + ], + [ + "Ġcom", + "fort" + ], + [ + "Ġ", + "ent" + ], + [ + "Ġre", + "p" + ], + [ + "Ġpers", + "on" + ], + [ + "Ġsmil", + "ing" + ], + [ + "ĠW", + "ith" + ], + [ + "Ġgent", + "ly" + ], + [ + "Ġpoin", + "ted" + ], + [ + "E", + "very" + ], + [ + "ow", + "ed" + ], + [ + "Ġsp", + "o" + ], + [ + "co", + "l" + ], + [ + "Ġtast", + "y" + ], + [ + "ul", + "ar" + ], + [ + "ĠJim", + "my" + ], + [ + "Ġpl", + "ane" + ], + [ + "st", + "ing" + ], + [ + "h", + "in" + ], + [ + "Ġh", + "oney" + ], + [ + "it", + "ch" + ], + [ + "Ġp", + "ill" + ], + [ + "Ġp", + "ass" + ], + [ + "itt", + "en" + ], + [ + "Ġm", + "atter" + ], + [ + "Ġad", + "m" + ], + [ + "Ġtast", + "ed" + ], + [ + "Ġsmell", + "ed" + ], + [ + "n", + "ic" + ], + [ + "o", + "ve" + ], + [ + "Ġe", + "ag" + ], + [ + "Ġsh", + "ining" + ], + [ + "He", + "y" + ], + [ + "Ġhop", + "e" + ], + [ + "Ġplac", + "es" + ], + [ + "Ġk", + "ick" + ], + [ + "ĠC", + "h" + ], + [ + "Ġpic", + "nic" + ], + [ + "Ġbott", + "le" + ], + [ + "Ġresp", + "ect" + ], + [ + "Ġb", + "lew" + ], + [ + "Ġapp", + "eared" + ], + [ + "Ġsu", + "pp" + ], + [ + "Tom", + "my" + ], + [ + "ĠC", + "ome" + ], + [ + "Ġsp", + "ider" + ], + [ + "Ġint", + "ere" + ], + [ + "Ġche", + "er" + ], + [ + "bo", + "ard" + ], + [ + "Ġwe", + "aring" + ], + [ + "Ġpres", + "ent" + ], + [ + "er", + "t" + ], + [ + "Ġm", + "ist" + ], + [ + "iz", + "e" + ], + [ + "Ġtrou", + "ble" + ], + [ + "Ġtri", + "p" + ], + [ + "Ġp", + "ract" + ], + [ + "Ġk", + "iss" + ], + [ + "Ġqu", + "est" + ], + [ + "o", + "nd" + ], + [ + "Ġas", + "h" + ], + [ + "Ġlo", + "ves" + ], + [ + "Ġtight", + "ly" + ], + [ + "g", + "g" + ], + [ + "ar", + "ge" + ], + [ + "Ġs", + "ur" + ], + [ + "Ġsp", + "read" + ], + [ + "a", + "ur" + ], + [ + "Ġw", + "ide" + ], + [ + "on", + "es" + ], + [ + "um", + "ber" + ], + [ + "Ġfe", + "ather" + ], + [ + "Ġspl", + "as" + ], + [ + "on", + "t" + ], + [ + "nd", + "p" + ], + [ + "Ġl", + "ater" + ], + [ + "Ġwho", + "le" + ], + [ + "Ġany", + "one" + ], + [ + "Ġbre", + "ad" + ], + [ + "M", + "olly" + ], + [ + "Ġd", + "es" + ], + [ + "Ġre", + "c" + ], + [ + "ell", + "a" + ], + [ + "Ġro", + "ad" + ], + [ + "I", + "n" + ], + [ + "Ġo", + "ce" + ], + [ + "Ġgi", + "ant" + ], + [ + "Ġoce", + "an" + ], + [ + "el", + "s" + ], + [ + "Ġpu", + "zz" + ], + [ + "Ġcook", + "ie" + ], + [ + "Ġp", + "an" + ], + [ + "Ġp", + "ath" + ], + [ + "Åĵ", + "I" + ], + [ + "Ġcon", + "fused" + ], + [ + "Ġscream", + "ed" + ], + [ + "Ġcloud", + "s" + ], + [ + "e", + "b" + ], + [ + "Ġimp", + "ress" + ], + [ + "Ġfr", + "ont" + ], + [ + "ĠG", + "o" + ], + [ + "Ġse", + "at" + ], + [ + "le", + "x" + ], + [ + "Ġl", + "ay" + ], + [ + "Ġwh", + "ich" + ], + [ + "l", + "a" + ], + [ + "Ġth", + "in" + ], + [ + "Ġte", + "ach" + ], + [ + "er", + "ly" + ], + [ + "Ġd", + "i" + ], + [ + "Ġansw", + "er" + ], + [ + "ung", + "le" + ], + [ + "Ġa", + "p" + ], + [ + "ss", + "ed" + ], + [ + "h", + "ile" + ], + [ + "u", + "b" + ], + [ + "Ġl", + "arge" + ], + [ + "Ġj", + "ungle" + ], + [ + "Ġprom", + "ise" + ], + [ + "Ġs", + "er" + ], + [ + "Ġb", + "er" + ], + [ + "Ġw", + "ild" + ], + [ + "Ġb", + "ark" + ], + [ + "Ġm", + "ach" + ], + [ + "oo", + "se" + ], + [ + "Ġbutt", + "on" + ], + [ + "ndp", + "a" + ], + [ + "G", + "ood" + ], + [ + "Ġmy", + "ster" + ], + [ + "Ġsn", + "ake" + ], + [ + "Ġvill", + "age" + ], + [ + "Ġh", + "or" + ], + [ + "ĠEm", + "ily" + ], + [ + "C", + "ome" + ], + [ + "Ġtoo", + "l" + ], + [ + "Ġput", + "s" + ], + [ + "Ġla", + "st" + ], + [ + "Ġim", + "ag" + ], + [ + "ĠJ", + "ake" + ], + [ + "L", + "ucy" + ], + [ + "Ġt", + "id" + ], + [ + "Ġb", + "an" + ], + [ + "le", + "br" + ], + [ + "Ġe", + "ventually" + ], + [ + "Ġsl", + "id" + ], + [ + "Ġwa", + "ve" + ], + [ + "Ġli", + "ve" + ], + [ + "Ġch", + "ased" + ], + [ + "ĠEvery", + "where" + ], + [ + "Ġce", + "lebr" + ], + [ + "Ġcray", + "ons" + ], + [ + "J", + "ust" + ], + [ + "Ġsp", + "ent" + ], + [ + "Ġwr", + "ite" + ], + [ + "Ġgi", + "ves" + ], + [ + "un", + "k" + ], + [ + "if", + "e" + ], + [ + "Ġdec", + "or" + ], + [ + "Ġfo", + "ot" + ], + [ + "Ġdel", + "ight" + ], + [ + "Ġso", + "l" + ], + [ + "Ġsand", + "w" + ], + [ + "Ġdir", + "t" + ], + [ + "ĠJ", + "enny" + ], + [ + "Ġtal", + "king" + ], + [ + "Ġtreat", + "s" + ], + [ + "Ġc", + "art" + ], + [ + "ch", + "ing" + ], + [ + "Ġwait", + "ing" + ], + [ + "Ġhid", + "ing" + ], + [ + "Ġsnow", + "man" + ], + [ + "Ġintere", + "sting" + ], + [ + "Ġh", + "on" + ], + [ + "on", + "y" + ], + [ + "Ġshe", + "lf" + ], + [ + "Ġtast", + "e" + ], + [ + "e", + "en" + ], + [ + "J", + "im" + ], + [ + "Ġst", + "ep" + ], + [ + "M", + "um" + ], + [ + "Ġhelp", + "ful" + ], + [ + "Ġfe", + "et" + ], + [ + "Ġsh", + "y" + ], + [ + "Ġgo", + "es" + ], + [ + "Ġv", + "al" + ], + [ + "Ġe", + "mb" + ], + [ + "Ġst", + "ood" + ], + [ + "Ġsh", + "r" + ], + [ + "Ġbuild", + "ing" + ], + [ + "Ġo", + "ven" + ], + [ + "Ġst", + "ret" + ], + [ + "Ġlove", + "ly" + ], + [ + "Ġduck", + "s" + ], + [ + "Ġcorn", + "er" + ], + [ + "Ġwas", + "h" + ], + [ + "ck", + "s" + ], + [ + "Ġsp", + "e" + ], + [ + "Ġpretend", + "ed" + ], + [ + "Ġroll", + "ed" + ], + [ + "Ġru", + "de" + ], + [ + "o", + "ld" + ], + [ + "Ġc", + "alm" + ], + [ + "Ġp", + "ile" + ], + [ + "Ġte", + "eth" + ], + [ + "Ġjump", + "ing" + ], + [ + "Ġz", + "oo" + ], + [ + "Ġpudd", + "le" + ], + [ + "Ġtid", + "y" + ], + [ + "M", + "ama" + ], + [ + "ig", + "er" + ], + [ + "ip", + "e" + ], + [ + "Ġbug", + "s" + ], + [ + "Ġash", + "amed" + ], + [ + "Ġb", + "at" + ], + [ + "in", + "a" + ], + [ + "ĠR", + "e" + ], + [ + "Ġo", + "b" + ], + [ + "Ġst", + "ra" + ], + [ + "ĠS", + "t" + ], + [ + "Ġwhist", + "le" + ], + [ + "Ġban", + "an" + ], + [ + "n", + "ot" + ], + [ + "Ġn", + "et" + ], + [ + "Ġforg", + "ive" + ], + [ + "Ġprin", + "ce" + ], + [ + "en", + "er" + ], + [ + "Ġsh", + "irt" + ], + [ + "Ġs", + "el" + ], + [ + "Ġc", + "oo" + ], + [ + "Ġemb", + "ar" + ], + [ + "ump", + "y" + ], + [ + "Ġspark", + "ly" + ], + [ + "Ġrem", + "ind" + ], + [ + "Ġs", + "ug" + ], + [ + "Ġc", + "ross" + ], + [ + "Ġm", + "ind" + ], + [ + "Ġcan", + "not" + ], + [ + "Ġsuc", + "cess" + ], + [ + "Ġembar", + "ra" + ], + [ + "S", + "oon" + ], + [ + "Ġm", + "ot" + ], + [ + "Ġsh", + "aring" + ], + [ + "Ġwor", + "m" + ], + [ + "Ġfo", + "ld" + ], + [ + "Ġpeace", + "ful" + ], + [ + "Ġw", + "ore" + ], + [ + "Ġm", + "ountain" + ], + [ + "Ġbl", + "ow" + ], + [ + "l", + "ice" + ], + [ + "Ġ", + "ign" + ], + [ + "Ġbe", + "ll" + ], + [ + "Ġfur", + "ry" + ], + [ + "Ġgra", + "b" + ], + [ + "al", + "s" + ], + [ + "omet", + "imes" + ], + [ + "Ġwhen", + "ever" + ], + [ + "Ġp", + "our" + ], + [ + "v", + "ous" + ], + [ + "ll", + "ie" + ], + [ + "Ġpu", + "sh" + ], + [ + "Ġbre", + "ath" + ], + [ + "Ġmach", + "ine" + ], + [ + "Ġsug", + "ar" + ], + [ + "Ġch", + "ick" + ], + [ + "Ġcreat", + "ive" + ], + [ + "S", + "t" + ], + [ + "n", + "ed" + ], + [ + "Ġb", + "al" + ], + [ + "Ġst", + "ir" + ], + [ + "Ġdo", + "lp" + ], + [ + "Ġtri", + "es" + ], + [ + "N", + "ow" + ], + [ + "Ġl", + "ad" + ], + [ + "ĠL", + "e" + ], + [ + "Ġm", + "ed" + ], + [ + "Ġp", + "ool" + ], + [ + "Ġg", + "ener" + ], + [ + "Ġsa", + "l" + ], + [ + "or", + "k" + ], + [ + "ul", + "t" + ], + [ + "Ġspr", + "ay" + ], + [ + "Ġsh", + "ip" + ], + [ + "i", + "an" + ], + [ + "Ġh", + "um" + ], + [ + "Ġhe", + "l" + ], + [ + "is", + "a" + ], + [ + "Ġsel", + "fish" + ], + [ + "Ġh", + "ar" + ], + [ + "Ġfa", + "ces" + ], + [ + "Ġmess", + "y" + ], + [ + "Ġ", + "'" + ], + [ + "Ġp", + "le" + ], + [ + "ĠS", + "ome" + ], + [ + "ĠH", + "ow" + ], + [ + "Ġgo", + "ld" + ], + [ + "Ġcho", + "col" + ], + [ + "P", + "lease" + ], + [ + "Ġminut", + "es" + ], + [ + "Ġpill", + "ow" + ], + [ + "M", + "y" + ], + [ + "a", + "king" + ], + [ + "Ġt", + "un" + ], + [ + "Ġm", + "issed" + ], + [ + "ig", + "hed" + ], + [ + "Ġv", + "an" + ], + [ + "Ġfix", + "ed" + ], + [ + "Ġfight", + "ing" + ], + [ + "o", + "in" + ], + [ + "Ġf", + "ig" + ], + [ + "Ġembarra", + "ssed" + ], + [ + "Ġme", + "ant" + ], + [ + "Ġdri", + "ve" + ], + [ + "Ġtom", + "orrow" + ], + [ + "Ġh", + "ours" + ], + [ + "Ġc", + "a" + ], + [ + "Ġn", + "er" + ], + [ + "Ġsp", + "icy" + ], + [ + "Ġknow", + "ing" + ], + [ + "ĠP", + "lease" + ], + [ + "Sara", + "h" + ], + [ + "Ġlad", + "der" + ], + [ + "it", + "al" + ], + [ + "Ġhor", + "se" + ], + [ + "at", + "o" + ], + [ + "Ġu", + "g" + ], + [ + "ĠJ", + "ust" + ], + [ + "Ġtr", + "ust" + ], + [ + "Ġho", + "sp" + ], + [ + "Ġug", + "ly" + ], + [ + "Ġhosp", + "ital" + ], + [ + "Ġche", + "w" + ], + [ + "Ġbed", + "room" + ], + [ + "Ġeas", + "y" + ], + [ + "Ġner", + "vous" + ], + [ + "Ġro", + "b" + ], + [ + "Ġthank", + "ful" + ], + [ + "Ġcarrot", + "s" + ], + [ + "D", + "ad" + ], + [ + "ell", + "y" + ], + [ + "Ġgre", + "w" + ], + [ + "Ġcol", + "our" + ], + [ + "Ġbar", + "ked" + ], + [ + "Ġcou", + "ch" + ], + [ + "Ġn", + "ut" + ], + [ + "Ġme", + "adow" + ], + [ + "Ġta", + "ught" + ], + [ + "Ġunderst", + "ood" + ], + [ + "Ġdisapp", + "oin" + ], + [ + "Ġp", + "as" + ], + [ + "an", + "o" + ], + [ + "Ġof", + "ten" + ], + [ + "Ġcl", + "ass" + ], + [ + "Ġpas", + "sed" + ], + [ + "Ġkn", + "ocked" + ], + [ + "Ġland", + "ed" + ], + [ + "Ġm", + "ic" + ], + [ + "ir", + "d" + ], + [ + "Ġfe", + "ar" + ], + [ + "Ġco", + "in" + ], + [ + "Ġmu", + "d" + ], + [ + "w", + "ork" + ], + [ + "Ġd", + "eter" + ], + [ + "Ġre", + "g" + ], + [ + "Ġreturn", + "ed" + ], + [ + "Ġdeter", + "m" + ], + [ + "Ġh", + "ur" + ], + [ + "Ġfin", + "ish" + ], + [ + "iz", + "z" + ], + [ + "Ġte", + "a" + ], + [ + "Ġsmo", + "oth" + ], + [ + "m", + "o" + ], + [ + "Ġst", + "uff" + ], + [ + "Ġmo", + "v" + ], + [ + "Ġgener", + "ous" + ], + [ + "T", + "o" + ], + [ + "Ġ", + "On" + ], + [ + "is", + "p" + ], + [ + "Ġdr", + "um" + ], + [ + "o", + "bby" + ], + [ + "Ġbubble", + "s" + ], + [ + "Ġrob", + "ot" + ], + [ + "Ġthe", + "se" + ], + [ + "Ġse", + "ed" + ], + [ + "Ġgl", + "ass" + ], + [ + "Ġdisappoin", + "ted" + ], + [ + "g", + "round" + ], + [ + "Ġr", + "ing" + ], + [ + "Ġc", + "ard" + ], + [ + "Ġhe", + "ars" + ], + [ + "Ġcar", + "ried" + ], + [ + "Ġmo", + "on" + ], + [ + "Ġchocol", + "ate" + ], + [ + "Ġlea", + "f" + ], + [ + "Ġsnack", + "s" + ], + [ + "Ġcir", + "c" + ], + [ + "Ġfan", + "cy" + ], + [ + "d", + "le" + ], + [ + "Ġsp", + "end" + ], + [ + "ĠA", + "nn" + ], + [ + "ust", + "r" + ], + [ + "Ġcra", + "b" + ], + [ + "Ġsong", + "s" + ], + [ + "p", + "ack" + ], + [ + "Ġp", + "ir" + ], + [ + "eci", + "ally" + ], + [ + "os", + "aur" + ], + [ + "Ġso", + "ld" + ], + [ + "Ġbr", + "idge" + ], + [ + "ĠE", + "ven" + ], + [ + "Ġsleep", + "y" + ], + [ + "Ġhon", + "est" + ], + [ + "Ġm", + "ir" + ], + [ + "ĠB", + "r" + ], + [ + "Ġpa", + "w" + ], + [ + "p", + "ecially" + ], + [ + "Ġa", + "v" + ], + [ + "ĠW", + "hy" + ], + [ + "iz", + "zy" + ], + [ + "Ġche", + "st" + ], + [ + "Ġwo", + "lf" + ], + [ + "Ġdin", + "osaur" + ], + [ + "Ġes", + "pecially" + ], + [ + "Ġm", + "ap" + ], + [ + "Ġg", + "as" + ], + [ + "or", + "y" + ], + [ + "Ġch", + "ange" + ], + [ + "Ġgr", + "oup" + ], + [ + "Ġgather", + "ed" + ], + [ + "Ġs", + "our" + ], + [ + "Ġpo", + "or" + ], + [ + "Ġspl", + "ash" + ], + [ + "Ġwa", + "ves" + ], + [ + "Ġfor", + "ward" + ], + [ + "Ġcl", + "own" + ], + [ + "Ġmyster", + "ious" + ], + [ + "r", + "or" + ], + [ + "Ġbo", + "dy" + ], + [ + "Ġfr", + "ustr" + ], + [ + "Ġcra", + "w" + ], + [ + "Ġf", + "ake" + ], + [ + "Ġp", + "in" + ], + [ + "Ġo", + "k" + ], + [ + "Ġr", + "ad" + ], + [ + "Ġwh", + "isp" + ], + [ + "Ġtw", + "irl" + ], + [ + "Ġstr", + "ing" + ], + [ + "Ġsmell", + "y" + ], + [ + "ĠTo", + "gether" + ], + [ + "s", + "ist" + ], + [ + "Ġc", + "and" + ], + [ + "Ġg", + "ate" + ], + [ + "Ġsc", + "ar" + ], + [ + "ĠK", + "itty" + ], + [ + "Ġneck", + "l" + ], + [ + "B", + "illy" + ], + [ + "Ġmot", + "or" + ], + [ + "y", + "ing" + ], + [ + "ot", + "e" + ], + [ + "Ġsh", + "ore" + ], + [ + "ur", + "se" + ], + [ + "Ġje", + "w" + ], + [ + "Ġwe", + "ak" + ], + [ + "Ġgr", + "umpy" + ], + [ + "Ġfing", + "er" + ], + [ + "Ġsa", + "ck" + ], + [ + "Ġk", + "itten" + ], + [ + "Ġpol", + "ite" + ], + [ + "Ġgl", + "ue" + ], + [ + "u", + "ce" + ], + [ + "Ġn", + "umber" + ], + [ + "air", + "s" + ], + [ + "Ġes", + "c" + ], + [ + "Ġmir", + "ror" + ], + [ + "Ġshe", + "ep" + ], + [ + "Ġv", + "ase" + ], + [ + "Ġplant", + "s" + ], + [ + "Ġber", + "ries" + ], + [ + "mo", + "st" + ], + [ + "B", + "e" + ], + [ + "d", + "en" + ], + [ + "Ġj", + "elly" + ], + [ + "ĠThe", + "ir" + ], + [ + "Ġe", + "mp" + ], + [ + "itt", + "er" + ], + [ + "The", + "re" + ], + [ + "Ġsc", + "r" + ], + [ + "Ġgr", + "ay" + ], + [ + "Ġpuzz", + "le" + ], + [ + "Ġneckl", + "ace" + ], + [ + "Ġm", + "aybe" + ], + [ + "Ġpl", + "ate" + ], + [ + "Ġal", + "most" + ], + [ + "us", + "ie" + ], + [ + "Ġfrustr", + "ated" + ], + [ + "Ġemp", + "ty" + ], + [ + "p", + "ort" + ], + [ + "or", + "ry" + ], + [ + "Ġth", + "ick" + ], + [ + "Ġj", + "am" + ], + [ + "Ġsk", + "ipped" + ], + [ + "ens", + "ive" + ], + [ + "ĠT", + "eddy" + ], + [ + "ag", + "ed" + ], + [ + "Ġclos", + "et" + ], + [ + "Ġhid", + "den" + ], + [ + "Ġexp", + "ensive" + ], + [ + "Ġdeterm", + "ined" + ], + [ + "Ġb", + "ored" + ], + [ + "Ġl", + "it" + ], + [ + "Ġd", + "rew" + ], + [ + "Ġle", + "gs" + ], + [ + "Ġho", + "pping" + ], + [ + "at", + "es" + ], + [ + "ĠS", + "ometimes" + ], + [ + "Ġstick", + "s" + ], + [ + "Ġon", + "es" + ], + [ + "ĠA", + "t" + ], + [ + "Ġbird", + "ie" + ], + [ + "ĠD", + "on" + ], + [ + "Ġpol", + "ice" + ], + [ + "Ġfruit", + "s" + ], + [ + "Ġt", + "ick" + ], + [ + "Ġa", + "unt" + ], + [ + "Ġw", + "is" + ], + [ + "Ġb", + "ake" + ], + [ + "Ġl", + "ine" + ], + [ + "ot", + "s" + ], + [ + "Ġst", + "one" + ], + [ + "Ġdog", + "s" + ], + [ + "au", + "l" + ], + [ + "Ġfig", + "ure" + ], + [ + "J", + "ane" + ], + [ + "Ġs", + "ight" + ], + [ + "Ġc", + "ries" + ], + [ + "ĠB", + "e" + ], + [ + "Ġmo", + "ving" + ], + [ + "Ġcont", + "ent" + ], + [ + "ĠBr", + "own" + ], + [ + "Ġsa", + "ved" + ], + [ + "Ġcl", + "oth" + ], + [ + "?\"", + "." + ], + [ + "bo", + "x" + ], + [ + "Ġbran", + "ches" + ], + [ + "Ġpack", + "ed" + ], + [ + "Ġpower", + "ful" + ], + [ + "Ġsplas", + "hed" + ], + [ + "Ġfe", + "ed" + ], + [ + "est", + "ed" + ], + [ + "Ġyell", + "ed" + ], + [ + "W", + "here" + ], + [ + "d", + "ge" + ], + [ + "k", + "in" + ], + [ + "Ġt", + "iger" + ], + [ + "Ġp", + "ay" + ], + [ + "id", + "dle" + ], + [ + "oo", + "p" + ], + [ + "ĠB", + "ella" + ], + [ + "Ġex", + "am" + ], + [ + "Ġta", + "ken" + ], + [ + "Ġcr", + "ane" + ], + [ + "Ġspo", + "il" + ], + [ + "'", + "d" + ], + [ + "Ġha", + "ng" + ], + [ + "Ġfl", + "ag" + ], + [ + "Ġansw", + "ered" + ], + [ + "Ġdisco", + "vered" + ], + [ + "ĠLe", + "o" + ], + [ + "u", + "sh" + ], + [ + "Ġsa", + "ve" + ], + [ + "Ġsee", + "k" + ], + [ + "Ġexc", + "ite" + ], + [ + "Ġpr", + "ay" + ], + [ + "Ġjo", + "g" + ], + [ + "Ġstand", + "ing" + ], + [ + "Ġdolp", + "hin" + ], + [ + "Ġj", + "ack" + ], + [ + "A", + "re" + ], + [ + "er", + "a" + ], + [ + "ra", + "g" + ], + [ + "Ġfl", + "ut" + ], + [ + "get", + "able" + ], + [ + "Ġb", + "oring" + ], + [ + "Ġlo", + "ck" + ], + [ + "Ġinc", + "red" + ], + [ + "Ġexcite", + "ment" + ], + [ + "Ġs", + "ighed" + ], + [ + "Ġw", + "ip" + ], + [ + "Ġf", + "is" + ], + [ + "Ġf", + "ill" + ], + [ + "Ġso", + "ap" + ], + [ + "ĠM", + "um" + ], + [ + "ol", + "a" + ], + [ + "Ġple", + "ased" + ], + [ + "Ġjack", + "et" + ], + [ + "l", + "ight" + ], + [ + "Ġs", + "ugg" + ], + [ + "Ġm", + "iddle" + ], + [ + "Ġe", + "ld" + ], + [ + "Ġwr", + "ote" + ], + [ + "Ġballoon", + "s" + ], + [ + "Ġorgan", + "ized" + ], + [ + "qu", + "e" + ], + [ + "Ġsail", + "ed" + ], + [ + "The", + "ir" + ], + [ + "Ġtell", + "s" + ], + [ + "Ġo", + "y" + ], + [ + "Ġst", + "ones" + ], + [ + "Ġbo", + "ard" + ], + [ + "Ġsp", + "un" + ], + [ + "Ġve", + "getable" + ], + [ + "l", + "ies" + ], + [ + "Ġ", + "ind" + ], + [ + "in", + "ess" + ], + [ + "Ġwor", + "king" + ], + [ + "l", + "ower" + ], + [ + "al", + "k" + ], + [ + "if", + "f" + ], + [ + "Ġpar", + "rot" + ], + [ + "St", + "op" + ], + [ + "Ġscar", + "f" + ], + [ + "Ġwa", + "gged" + ], + [ + "Ġcl", + "ap" + ], + [ + "Ġme", + "asure" + ], + [ + "as", + "ing" + ], + [ + "Ġte", + "ars" + ], + [ + "bo", + "dy" + ], + [ + "Ġcomfort", + "able" + ], + [ + "Ġle", + "g" + ], + [ + "Ġsw", + "an" + ], + [ + "Ġtr", + "ue" + ], + [ + "Ġfr", + "ight" + ], + [ + "Ġdr", + "op" + ], + [ + "Ġsmo", + "ke" + ], + [ + "Ġfeather", + "s" + ], + [ + "Ġincred", + "ible" + ], + [ + "g", + "gs" + ], + [ + "u", + "al" + ], + [ + "Ġt", + "ie" + ], + [ + "Ġc", + "ap" + ], + [ + "Ġlo", + "se" + ], + [ + "Ġpu", + "s" + ], + [ + "Ġeag", + "er" + ], + [ + "Ġmov", + "ie" + ], + [ + "f", + "ic" + ], + [ + "Ġc", + "op" + ], + [ + "Ġe", + "ggs" + ], + [ + "oo", + "f" + ], + [ + "Ġr", + "id" + ], + [ + "Ġco", + "vered" + ], + [ + "Ġf", + "il" + ], + [ + "Ġle", + "m" + ], + [ + "Ġgra", + "p" + ], + [ + "Ġdisapp", + "eared" + ], + [ + "Ġrec", + "ord" + ], + [ + "ĠT", + "V" + ], + [ + "ĠB", + "l" + ], + [ + "Ġex", + "t" + ], + [ + "Ġor", + "d" + ], + [ + "Ġgigg", + "led" + ], + [ + "c", + "orn" + ], + [ + "Ġp", + "ri" + ], + [ + "Ġha", + "m" + ], + [ + "Ġr", + "are" + ], + [ + "Ġyour", + "self" + ], + [ + "Ġen", + "g" + ], + [ + "Ġwh", + "ale" + ], + [ + "Ġkn", + "ife" + ], + [ + "ĠL", + "isa" + ], + [ + "Ġte", + "am" + ], + [ + "ĠJohn", + "ny" + ], + [ + "z", + "ed" + ], + [ + "Ġt", + "est" + ], + [ + "Ġb", + "one" + ], + [ + "Ġd", + "izzy" + ], + [ + "Ġli", + "br" + ], + [ + "Ġwra", + "p" + ], + [ + "Ġquest", + "ions" + ], + [ + "i", + "v" + ], + [ + "Ġl", + "ying" + ], + [ + "Ġst", + "amp" + ], + [ + "Ġli", + "cked" + ], + [ + "Ġno", + "isy" + ], + [ + "Ġkind", + "ness" + ], + [ + "Ġdif", + "fic" + ], + [ + "Ġdraw", + "ing" + ], + [ + "app", + "ing" + ], + [ + "Jim", + "my" + ], + [ + "Ġstra", + "w" + ], + [ + "Ġb", + "urn" + ], + [ + "Ġwa", + "gon" + ], + [ + "at", + "ely" + ], + [ + "Ġgr", + "own" + ], + [ + "Ġrock", + "et" + ], + [ + "Ġpop", + "corn" + ], + [ + "S", + "ally" + ], + [ + "Ġb", + "atter" + ], + [ + "Ġwa", + "nd" + ], + [ + "Ġan", + "c" + ], + [ + "ĠE", + "nd" + ], + [ + "Ġsw", + "ord" + ], + [ + "Ġhappen", + "ing" + ], + [ + "Ġlift", + "ed" + ], + [ + "Ġbal", + "ance" + ], + [ + "Ġext", + "ra" + ], + [ + "S", + "orry" + ], + [ + "h", + "a" + ], + [ + "Ġwa", + "nder" + ], + [ + "Ġf", + "at" + ], + [ + "ĠM", + "ark" + ], + [ + "ĠE", + "lla" + ], + [ + "Ġbec", + "ome" + ], + [ + "Ġtr", + "ack" + ], + [ + "Ġsmall", + "er" + ], + [ + "Ġmist", + "ake" + ], + [ + "Ġto", + "ast" + ], + [ + "at", + "ing" + ], + [ + "Ġman", + "aged" + ], + [ + "sel", + "ves" + ], + [ + "ĠP", + "eter" + ], + [ + "Ġbath", + "room" + ], + [ + "Ġhar", + "m" + ], + [ + "Ġdiffic", + "ult" + ], + [ + "Ġtw", + "ist" + ], + [ + "Ġoff", + "ered" + ], + [ + "Ġru", + "ined" + ], + [ + "Ġmat", + "ch" + ], + [ + "Ġanc", + "ient" + ], + [ + "Ġg", + "ar" + ], + [ + "ill", + "i" + ], + [ + "Ġdist", + "ance" + ], + [ + "Ġplay", + "ground" + ], + [ + "Ġla", + "zy" + ], + [ + "Ġdan", + "cing" + ], + [ + "Ġadvent", + "ur" + ], + [ + "a", + "f" + ], + [ + "ĠB", + "a" + ], + [ + "Ġcall", + "ing" + ], + [ + "Ġclean", + "ed" + ], + [ + "Ġcr", + "own" + ], + [ + "orm", + "al" + ], + [ + "Ġon", + "ce" + ], + [ + "Ġbo", + "ss" + ], + [ + "Ġse", + "ll" + ], + [ + "Ġwalk", + "s" + ], + [ + "Ġfright", + "ened" + ], + [ + "er", + "ri" + ], + [ + "Ġm", + "ask" + ], + [ + "Ġfl", + "uffy" + ], + [ + "em", + "o" + ], + [ + "Ġ", + "Z" + ], + [ + "Ġf", + "rag" + ], + [ + "Ġd", + "est" + ], + [ + "Ġn", + "ormal" + ], + [ + "Ġy", + "og" + ], + [ + "Ġpick", + "s" + ], + [ + "Ġey", + "e" + ], + [ + "Ġdr", + "ank" + ], + [ + "Ġcir", + "cle" + ], + [ + "Ġadventur", + "ous" + ], + [ + "S", + "pot" + ], + [ + "Ġt", + "ent" + ], + [ + "Ġp", + "ress" + ], + [ + "Ġk", + "issed" + ], + [ + "ic", + "t" + ], + [ + "Ġsc", + "iss" + ], + [ + "Ġget", + "s" + ], + [ + "Ġtr", + "unk" + ], + [ + "Ġche", + "ck" + ], + [ + "Ġenjoy", + "ing" + ], + [ + "Ġshell", + "s" + ], + [ + "Ġeld", + "erly" + ], + [ + "Ġsciss", + "ors" + ], + [ + "Ġt", + "erri" + ], + [ + "Ġto", + "ug" + ], + [ + "ag", + "es" + ], + [ + "Ġste", + "al" + ], + [ + "M", + "e" + ], + [ + "re", + "ady" + ], + [ + "et", + "e" + ], + [ + "Ġal", + "ready" + ], + [ + "ĠL", + "ittle" + ], + [ + "Ġju", + "icy" + ], + [ + "Ġstu", + "dy" + ], + [ + "Ġh", + "orn" + ], + [ + "er", + "p" + ], + [ + "Ġp", + "izz" + ], + [ + "Ġall", + "ig" + ], + [ + "Ġloud", + "er" + ], + [ + "Ġtow", + "el" + ], + [ + "Ġcirc", + "les" + ], + [ + "Ġle", + "ad" + ], + [ + "Ġro", + "ar" + ], + [ + "Ġro", + "se" + ], + [ + "Ġsl", + "ipped" + ], + [ + "Ġte", + "le" + ], + [ + "Ġfam", + "ous" + ], + [ + "Ġar", + "g" + ], + [ + "Ġcheer", + "ful" + ], + [ + "ĠRe", + "x" + ], + [ + "Ġtoug", + "h" + ], + [ + "i", + "pper" + ], + [ + "Ġb", + "arn" + ], + [ + "Ġf", + "ool" + ], + [ + "Ġd", + "ig" + ], + [ + "Ġd", + "ish" + ], + [ + "Ġagain", + "st" + ], + [ + "Ġde", + "er" + ], + [ + "Ġsweet", + "ie" + ], + [ + "Ġeng", + "ine" + ], + [ + "Ġ", + "ed" + ], + [ + "Ġf", + "ather" + ], + [ + "Ġco", + "ver" + ], + [ + "Ġyour", + "s" + ], + [ + "Ġneed", + "s" + ], + [ + "Jo", + "e" + ], + [ + "Ġbag", + "s" + ], + [ + "Ġspin", + "ning" + ], + [ + "S", + "ue" + ], + [ + "Ġa", + "w" + ], + [ + "an", + "ged" + ], + [ + "am", + "ond" + ], + [ + "Ġy", + "arn" + ], + [ + "op", + "h" + ], + [ + "Ġgi", + "ving" + ], + [ + "Ġdi", + "amond" + ], + [ + "Ġsandw", + "ich" + ], + [ + "Ġind", + "epend" + ], + [ + "Ġfrag", + "ile" + ], + [ + "W", + "ho" + ], + [ + "ce", + "pt" + ], + [ + "Ġno", + "sy" + ], + [ + "Ġtw", + "ig" + ], + [ + "Ġhurt", + "s" + ], + [ + "Ġsugg", + "ested" + ], + [ + "Ġlem", + "on" + ], + [ + "i", + "est" + ], + [ + "in", + "ation" + ], + [ + "or", + "ge" + ], + [ + "Ġno", + "d" + ], + [ + "Ġbl", + "oom" + ], + [ + "Ġho", + "se" + ], + [ + "Ġbox", + "es" + ], + [ + "Ġreal", + "ised" + ], + [ + "Ġfr", + "own" + ], + [ + "ac", + "he" + ], + [ + "Ġrel", + "ax" + ], + [ + "Ġham", + "mer" + ], + [ + "Ġa", + "dded" + ], + [ + "Ġwe", + "ird" + ], + [ + "Ġmo", + "d" + ], + [ + "Ġcr", + "ack" + ], + [ + "Ġallig", + "ator" + ], + [ + "A", + "my" + ], + [ + "G", + "ra" + ], + [ + "o", + "ot" + ], + [ + "Ġsh", + "ake" + ], + [ + "ra", + "ge" + ], + [ + "Ġlearn", + "ing" + ], + [ + "Ġvegetable", + "s" + ], + [ + "Ġa", + "dd" + ], + [ + "Ġm", + "elt" + ], + [ + "ĠS", + "usie" + ], + [ + "el", + "p" + ], + [ + "ĠB", + "obby" + ], + [ + "Ġtra", + "p" + ], + [ + "Ġcoll", + "ar" + ], + [ + "Ġstu", + "bb" + ], + [ + "i", + "que" + ], + [ + "Ġb", + "orrow" + ], + [ + "Ġp", + "ump" + ], + [ + "Ġloud", + "ly" + ], + [ + "Ġsear", + "ch" + ], + [ + "Ġchick", + "en" + ], + [ + "Ġpizz", + "a" + ], + [ + "Ġindepend", + "ent" + ], + [ + "Ġstubb", + "orn" + ], + [ + "Ġre", + "f" + ], + [ + "Ġas", + "king" + ], + [ + "Ġbr", + "illi" + ], + [ + "Ġdec", + "ide" + ], + [ + "Ġfl", + "ash" + ], + [ + "Ġcat", + "erp" + ], + [ + "Ġqu", + "een" + ], + [ + "Ġsn", + "ee" + ], + [ + "ĠR", + "ose" + ], + [ + "Ġinv", + "ited" + ], + [ + "Ġstraw", + "ber" + ], + [ + "Ġfool", + "ish" + ], + [ + "Ġcaterp", + "ill" + ], + [ + "g", + "y" + ], + [ + "Ġt", + "ur" + ], + [ + "Ġd", + "epend" + ], + [ + "Ġg", + "uit" + ], + [ + "Ġst", + "ri" + ], + [ + "Ġkn", + "ight" + ], + [ + "Ġboy", + "s" + ], + [ + "Ġpe", + "e" + ], + [ + "Ġcom", + "b" + ], + [ + "Ġcle", + "ar" + ], + [ + "Ġun", + "ique" + ], + [ + "Ġbutt", + "ons" + ], + [ + "ĠTweet", + "y" + ], + [ + "Ġcolour", + "ful" + ], + [ + "Ġesc", + "ape" + ], + [ + "Ġbrilli", + "ant" + ], + [ + "Ġr", + "ich" + ], + [ + "ĠB", + "u" + ], + [ + "Ġsay", + "ing" + ], + [ + "Ġro", + "de" + ], + [ + "Ġsk", + "in" + ], + [ + "Ġgar", + "age" + ], + [ + "H", + "elp" + ], + [ + "d", + "om" + ], + [ + "Ġm", + "ummy" + ], + [ + "Ġbe", + "ak" + ], + [ + "ur", + "ed" + ], + [ + "a", + "w" + ], + [ + "e", + "orge" + ], + [ + "r", + "op" + ], + [ + "ed", + "ient" + ], + [ + "Ġun", + "us" + ], + [ + "Ġoy", + "ster" + ], + [ + "Ġtur", + "key" + ], + [ + "Ġ", + "OK" + ], + [ + "Ġf", + "ountain" + ], + [ + "Ġp", + "atter" + ], + [ + "Ġwe", + "ek" + ], + [ + "Ġch", + "ase" + ], + [ + "Ġru", + "bb" + ], + [ + "Ġcount", + "ed" + ], + [ + "Ġreg", + "ular" + ], + [ + "Ġwip", + "ed" + ], + [ + "Ġguit", + "ar" + ], + [ + "Ġunus", + "ual" + ], + [ + "l", + "and" + ], + [ + "Ġu", + "mb" + ], + [ + "Ġfor", + "k" + ], + [ + "Ġshow", + "s" + ], + [ + "Ġlight", + "s" + ], + [ + "Ġwall", + "s" + ], + [ + "Ġdream", + "ed" + ], + [ + "Ġbanan", + "a" + ], + [ + "Ġsuccess", + "ful" + ], + [ + "Ġjew", + "el" + ], + [ + "Ġpatter", + "n" + ], + [ + "Ġumb", + "re" + ], + [ + "c", + "il" + ], + [ + "r", + "ay" + ], + [ + "in", + "al" + ], + [ + "Ġl", + "ow" + ], + [ + "im", + "ed" + ], + [ + "Ġsw", + "e" + ], + [ + "Ġcry", + "st" + ], + [ + "Ġpen", + "cil" + ], + [ + "Ġsaf", + "ely" + ], + [ + "Ġtool", + "s" + ], + [ + "W", + "ell" + ], + [ + "Ġt", + "ape" + ], + [ + "Ġst", + "ream" + ], + [ + "ce", + "ed" + ], + [ + "Ġre", + "sc" + ], + [ + "Ġab", + "ove" + ], + [ + "Ġhard", + "er" + ], + [ + "Ġgu", + "ess" + ], + [ + "Ġbott", + "om" + ], + [ + "Ġmark", + "et" + ], + [ + "Ġimpress", + "ed" + ], + [ + "Ġwa", + "ff" + ], + [ + "an", + "ut" + ], + [ + "ig", + "inal" + ], + [ + "Ġnot", + "ice" + ], + [ + "ill", + "a" + ], + [ + "Ġno", + "ds" + ], + [ + "Ġpe", + "anut" + ], + [ + "ol", + "og" + ], + [ + "Ġpat", + "ch" + ], + [ + "Ġcraw", + "led" + ], + [ + "Ġcaterpill", + "ar" + ], + [ + "n", + "a" + ], + [ + "Ġt", + "ank" + ], + [ + "Ġt", + "ask" + ], + [ + "Ġt", + "ire" + ], + [ + "Ġg", + "un" + ], + [ + "Ġlo", + "ad" + ], + [ + "Ġcomp", + "et" + ], + [ + "umb", + "led" + ], + [ + "Ġap", + "olog" + ], + [ + "Ġsol", + "ve" + ], + [ + "H", + "ow" + ], + [ + "M", + "ummy" + ], + [ + "f", + "ast" + ], + [ + "h", + "ip" + ], + [ + "Ġ", + "No" + ], + [ + "Ġb", + "itter" + ], + [ + "ar", + "l" + ], + [ + "Ġe", + "ars" + ], + [ + "Ġsh", + "one" + ], + [ + "Ġbr", + "ush" + ], + [ + "ĠF", + "in" + ], + [ + "Ġor", + "iginal" + ], + [ + "Ġfi", + "er" + ], + [ + "Ġgl", + "ow" + ], + [ + "ĠN", + "emo" + ], + [ + "Ġbreak", + "fast" + ], + [ + "Ġadm", + "ired" + ], + [ + "ĠBl", + "ue" + ], + [ + "Ġed", + "ge" + ], + [ + "Ġdepend", + "able" + ], + [ + "u", + "nd" + ], + [ + "ut", + "h" + ], + [ + "ĠI", + "f" + ], + [ + "Ġcl", + "ay" + ], + [ + "Ġstrong", + "er" + ], + [ + "Ġboss", + "y" + ], + [ + "Ġc", + "udd" + ], + [ + "as", + "er" + ], + [ + "Ġgra", + "ndpa" + ], + [ + "Ġknow", + "s" + ], + [ + "Ġpick", + "ing" + ], + [ + "Ġjo", + "ke" + ], + [ + "Ġboat", + "s" + ], + [ + "Ġba", + "ld" + ], + [ + "Ġmed", + "ic" + ], + [ + "Ġscr", + "at" + ], + [ + "Ġc", + "one" + ], + [ + "Ġp", + "enny" + ], + [ + "Ġle", + "an" + ], + [ + "Ġra", + "ft" + ], + [ + "ugg", + "led" + ], + [ + "Ġsuc", + "ceed" + ], + [ + "Ġob", + "edient" + ], + [ + "Ġhum", + "ble" + ], + [ + "Ġlibr", + "ary" + ], + [ + "R", + "e" + ], + [ + "he", + "ad" + ], + [ + "Ġp", + "h" + ], + [ + "Ġn", + "umb" + ], + [ + "Ġsa", + "uce" + ], + [ + "ĠB", + "ear" + ], + [ + "Ġtr", + "ay" + ], + [ + "Ġproud", + "ly" + ], + [ + "He", + "re" + ], + [ + "Ġsound", + "ed" + ], + [ + "Ġsqu", + "are" + ], + [ + "Ġterri", + "ble" + ], + [ + "Ġcryst", + "al" + ], + [ + "Ġfier", + "ce" + ], + [ + "J", + "ill" + ], + [ + "Ġw", + "itch" + ], + [ + "in", + "ary" + ], + [ + "Ġhapp", + "iness" + ], + [ + "ĠL", + "ola" + ], + [ + "Ġhand", + "ed" + ], + [ + "Ġfr", + "idge" + ], + [ + "f", + "a" + ], + [ + "Ġr", + "ough" + ], + [ + "ent", + "ion" + ], + [ + "Ġch", + "anged" + ], + [ + "Ġjo", + "ined" + ], + [ + "Ġra", + "ven" + ], + [ + "Ġmight", + "y" + ], + [ + "Ġcho", + "se" + ], + [ + "Ġwhe", + "el" + ], + [ + "Ġumbre", + "lla" + ], + [ + "Ġt", + "ag" + ], + [ + "Ġc", + "ri" + ], + [ + "Ġo", + "w" + ], + [ + "ar", + "ry" + ], + [ + "im", + "p" + ], + [ + "ĠB", + "ill" + ], + [ + "Ġhelp", + "s" + ], + [ + "Ġfind", + "ing" + ], + [ + "Ġclean", + "ing" + ], + [ + "Ġsqu", + "ee" + ], + [ + "Ġfire", + "work" + ], + [ + "Ġmedic", + "ine" + ], + [ + "a", + "zz" + ], + [ + "m", + "et" + ], + [ + "Ġplay", + "ful" + ], + [ + "Ġlo", + "cked" + ], + [ + "Ġgo", + "at" + ], + [ + "Ġac", + "cept" + ], + [ + "Ġcomp", + "ass" + ], + [ + "Ġsur", + "f" + ], + [ + "Ġcompet", + "it" + ], + [ + "e", + "xt" + ], + [ + "k", + "n" + ], + [ + "Ġs", + "y" + ], + [ + "Ġs", + "il" + ], + [ + "Ġbe", + "et" + ], + [ + "ĠM", + "ar" + ], + [ + "ur", + "ing" + ], + [ + "Ġch", + "ir" + ], + [ + "ĠE", + "llie" + ], + [ + "Ġball", + "s" + ], + [ + "Åĵ", + "Let" + ], + [ + "Ġbutterf", + "lies" + ], + [ + "Ġgu", + "ard" + ], + [ + "Ġcray", + "on" + ], + [ + "Ġdisco", + "ver" + ], + [ + "Ġpar", + "ade" + ], + [ + "Ġmix", + "ed" + ], + [ + "Ġval", + "u" + ], + [ + "Ġhel", + "met" + ], + [ + "ĠBrown", + "ie" + ], + [ + "E", + "m" + ], + [ + "j", + "ect" + ], + [ + "s", + "qu" + ], + [ + "u", + "d" + ], + [ + "Ġg", + "um" + ], + [ + "at", + "ure" + ], + [ + "Ġis", + "land" + ], + [ + "Ġco", + "st" + ], + [ + "Ġmo", + "le" + ], + [ + "Ġnice", + "ly" + ], + [ + "Ġkind", + "s" + ], + [ + "Ġra", + "ced" + ], + [ + "Ġsto", + "ve" + ], + [ + "Ġlaugh", + "s" + ], + [ + "Ġmar", + "ch" + ], + [ + "Ġfall", + "s" + ], + [ + "Ġpers", + "ist" + ], + [ + "ĠBa", + "by" + ], + [ + "oph", + "ie" + ], + [ + "A", + "lice" + ], + [ + "i", + "ence" + ], + [ + "it", + "o" + ], + [ + "Ġc", + "age" + ], + [ + "Ġg", + "oose" + ], + [ + "Ġco", + "ins" + ], + [ + "Ġtr", + "ump" + ], + [ + "Ġmo", + "squ" + ], + [ + "Ġser", + "ious" + ], + [ + "Ġmosqu", + "ito" + ], + [ + "Ġf", + "lex" + ], + [ + "Ġc", + "ro" + ], + [ + "el", + "on" + ], + [ + "Ġexc", + "la" + ], + [ + "Ġch", + "oose" + ], + [ + "Ġexpl", + "ored" + ], + [ + "Ġun", + "l" + ], + [ + "Ġbrother", + "s" + ], + [ + "Ġgu", + "il" + ], + [ + "Ġnumb", + "ers" + ], + [ + "Ġtrump", + "et" + ], + [ + "Ġ", + "icy" + ], + [ + "Ġmom", + "s" + ], + [ + "ie", + "w" + ], + [ + "Ġv", + "iew" + ], + [ + "Ġen", + "orm" + ], + [ + "Ġfire", + "man" + ], + [ + "ĠMr", + "s" + ], + [ + "Ġpract", + "ice" + ], + [ + "Ġenorm", + "ous" + ], + [ + "w", + "ards" + ], + [ + "Ġs", + "up" + ], + [ + "Ġvo", + "lc" + ], + [ + "Ġmean", + "s" + ], + [ + "Ġpoin", + "ting" + ], + [ + "Ġvalu", + "able" + ], + [ + "Ġflex", + "ible" + ], + [ + "Ġexcla", + "imed" + ], + [ + "Ġguil", + "ty" + ], + [ + "y", + "al" + ], + [ + "Ġc", + "am" + ], + [ + "Ġp", + "ipe" + ], + [ + "is", + "ion" + ], + [ + "ĠM", + "y" + ], + [ + "Ġuse", + "ful" + ], + [ + "Ġfav", + "our" + ], + [ + "Ġcra", + "zy" + ], + [ + "Ġph", + "ot" + ], + [ + "kn", + "own" + ], + [ + "Ġvolc", + "ano" + ], + [ + "Ġwa", + "ke" + ], + [ + "Ġc", + "ity" + ], + [ + "om", + "ed" + ], + [ + "ll", + "ip" + ], + [ + "Ġfor", + "th" + ], + [ + "Ġlo", + "llip" + ], + [ + "Ġse", + "al" + ], + [ + "Ġall", + "owed" + ], + [ + "Ġsc", + "are" + ], + [ + "Ġbright", + "ly" + ], + [ + "Ġmar", + "ble" + ], + [ + "Ġpop", + "ular" + ], + [ + "Ġflash", + "light" + ], + [ + "Ġcompass", + "ion" + ], + [ + "Ġlollip", + "op" + ], + [ + "Ġ", + "U" + ], + [ + "Ġ", + "ing" + ], + [ + "Ġh", + "unt" + ], + [ + "Ġd", + "ull" + ], + [ + "er", + "ry" + ], + [ + "Ġst", + "at" + ], + [ + "Ġshe", + "l" + ], + [ + "!\"", + "." + ], + [ + "Ġun", + "known" + ], + [ + "Ġhat", + "s" + ], + [ + "Ġsho", + "ot" + ], + [ + "red", + "ient" + ], + [ + "Ġbelong", + "ed" + ], + [ + "Ġfavour", + "ite" + ], + [ + "Ġing", + "redient" + ], + [ + "Ġp", + "il" + ], + [ + "Ġlo", + "yal" + ], + [ + "Ġbra", + "ce" + ], + [ + "ĠWhen", + "ever" + ], + [ + "Ġdream", + "s" + ], + [ + "Ġkick", + "ed" + ], + [ + "Ġseed", + "s" + ], + [ + "Ġtwirl", + "ed" + ], + [ + "Ġsup", + "er" + ], + [ + "t", + "ime" + ], + [ + "Ġf", + "ine" + ], + [ + "Ġbe", + "es" + ], + [ + "Ġso", + "fa" + ], + [ + "Ġsh", + "ark" + ], + [ + "ĠM", + "ike" + ], + [ + "Ġan", + "x" + ], + [ + "Ġsad", + "ly" + ], + [ + "Ġshow", + "ing" + ], + [ + "Ġway", + "s" + ], + [ + "Ġtruck", + "s" + ], + [ + "Ġdelic", + "ate" + ], + [ + "ation", + "s" + ], + [ + "Ġrep", + "e" + ], + [ + "Ġimpress", + "ive" + ], + [ + "Ġpour", + "ed" + ], + [ + "Ġpir", + "ate" + ], + [ + "Ġwhisp", + "ered" + ], + [ + "Ġharm", + "less" + ], + [ + "a", + "ff" + ], + [ + "Ġt", + "urt" + ], + [ + "Ġt", + "ied" + ], + [ + "ar", + "oo" + ], + [ + "ad", + "o" + ], + [ + "art", + "h" + ], + [ + "Ġgra", + "ce" + ], + [ + "Ġhand", + "le" + ], + [ + "Ġen", + "c" + ], + [ + "Ġdel", + "iver" + ], + [ + "Ben", + "ny" + ], + [ + "Ġru", + "shed" + ], + [ + "Ġus", + "ing" + ], + [ + "ang", + "aroo" + ], + [ + "Ġshap", + "e" + ], + [ + "Ġbus", + "hes" + ], + [ + "Ġpersist", + "ent" + ], + [ + "Ġg", + "or" + ], + [ + "ĠS", + "p" + ], + [ + "Ġk", + "angaroo" + ], + [ + "Ġan", + "g" + ], + [ + "Ġoff", + "ice" + ], + [ + "Ġanx", + "ious" + ], + [ + "G", + "o" + ], + [ + "al", + "ous" + ], + [ + "Ġsp", + "ell" + ], + [ + "ĠW", + "hile" + ], + [ + "ia", + "ble" + ], + [ + "Ġpot", + "ato" + ], + [ + "Ġje", + "alous" + ], + [ + "Ġwra", + "pped" + ], + [ + "Ġf", + "our" + ], + [ + "Ġc", + "urt" + ], + [ + "Ġlo", + "g" + ], + [ + "Ġco", + "al" + ], + [ + "Ġrel", + "iable" + ], + [ + "Ġcurt", + "ain" + ], + [ + "D", + "aisy" + ], + [ + "Ġs", + "udden" + ], + [ + "mb", + "ol" + ], + [ + "Åĵ", + "Yes" + ], + [ + "ac", + "hes" + ], + [ + "ĠP", + "e" + ], + [ + "Ġpain", + "ted" + ], + [ + "ĠTo", + "by" + ], + [ + "Ġce", + "re" + ], + [ + "g", + "u" + ], + [ + "Ġt", + "ummy" + ], + [ + "Ġth", + "ose" + ], + [ + "ous", + "es" + ], + [ + "udd", + "y" + ], + [ + "Ġcu", + "sh" + ], + [ + "Ġmet", + "al" + ], + [ + "Ġsy", + "mbol" + ], + [ + "Ġbeet", + "le" + ], + [ + "Ġcam", + "era" + ], + [ + "Ġh", + "ouses" + ], + [ + "il", + "t" + ], + [ + "Ġcelebr", + "ate" + ], + [ + "Ġdelight", + "ed" + ], + [ + "Ġgor", + "illa" + ], + [ + "Ġto", + "oth" + ], + [ + "Ġth", + "irst" + ], + [ + "ke", + "ep" + ], + [ + "Ġsh", + "iver" + ], + [ + "Ġre", + "ward" + ], + [ + "Ġsp", + "ace" + ], + [ + "Ġtra", + "vel" + ], + [ + "Ġru", + "bbed" + ], + [ + "Ġprin", + "t" + ], + [ + "Ġdri", + "ving" + ], + [ + "Ġmar", + "ry" + ], + [ + "Ġwar", + "ned" + ], + [ + "Ġsold", + "ier" + ], + [ + "Ġmotor", + "cy" + ], + [ + "c", + "ut" + ], + [ + "Ġs", + "on" + ], + [ + "Ġto", + "r" + ], + [ + "ar", + "lie" + ], + [ + "Ġher", + "o" + ], + [ + "ie", + "f" + ], + [ + "Ġcar", + "p" + ], + [ + "Ġexcited", + "ly" + ], + [ + "Ġte", + "mp" + ], + [ + "ĠP", + "ete" + ], + [ + "ĠBob", + "o" + ], + [ + "co", + "d" + ], + [ + "Ġhun", + "g" + ], + [ + "Ġgrow", + "ing" + ], + [ + "up", + "id" + ], + [ + "Ġthin", + "king" + ], + [ + "Ġtun", + "n" + ], + [ + "W", + "hile" + ], + [ + "c", + "u" + ], + [ + "i", + "pp" + ], + [ + "Ġh", + "ay" + ], + [ + "Ġy", + "et" + ], + [ + "Ġv", + "ine" + ], + [ + "Ġac", + "orn" + ], + [ + "icy", + "cle" + ], + [ + "Ġpre", + "p" + ], + [ + "Ġremind", + "ed" + ], + [ + "Ġgas", + "ped" + ], + [ + "Ġflut", + "e" + ], + [ + "Ġord", + "inary" + ], + [ + "Ġcro", + "cod" + ], + [ + "Ġa", + "mb" + ], + [ + "Ġc", + "ur" + ], + [ + "ĠB", + "et" + ], + [ + "ĠM", + "andy" + ], + [ + "Ġpu", + "p" + ], + [ + "Ġcreature", + "s" + ], + [ + "Ġhur", + "ry" + ], + [ + "Ġspoil", + "ed" + ], + [ + "Ġt", + "imes" + ], + [ + "Ġm", + "is" + ], + [ + "Ġr", + "ice" + ], + [ + "Ġst", + "upid" + ], + [ + "Ġsh", + "ine" + ], + [ + "Ġre", + "sist" + ], + [ + "ug", + "ged" + ], + [ + "Ġla", + "w" + ], + [ + "Ġsk", + "ull" + ], + [ + "co", + "a" + ], + [ + "Ġmiss", + "ing" + ], + [ + "Ġwhe", + "at" + ], + [ + "Ġsupp", + "ort" + ], + [ + "Ġingredient", + "s" + ], + [ + "a", + "id" + ], + [ + "l", + "oo" + ], + [ + "Ġc", + "ase" + ], + [ + "er", + "y" + ], + [ + "Ġwas", + "hed" + ], + [ + "Ġdo", + "ve" + ], + [ + "Ġsp", + "illed" + ], + [ + "Ġsc", + "oot" + ], + [ + "Ġme", + "al" + ], + [ + "Ġmu", + "le" + ], + [ + "ĠD", + "ucky" + ], + [ + "Ġmo", + "der" + ], + [ + "ars", + "h" + ], + [ + "Åĵ", + "What" + ], + [ + "Ġtri", + "ck" + ], + [ + "Ġset", + "t" + ], + [ + "ĠTweet", + "ie" + ], + [ + "Ġboun", + "ce" + ], + [ + "Ġfil", + "thy" + ], + [ + "Ġb", + "ase" + ], + [ + "Ġh", + "oop" + ], + [ + "Ġf", + "re" + ], + [ + "Ġst", + "umbled" + ], + [ + "Ġso", + "ar" + ], + [ + "Ġwe", + "al" + ], + [ + "Ġsee", + "m" + ], + [ + "Ġwor", + "d" + ], + [ + "Ġla", + "b" + ], + [ + "ĠD", + "ave" + ], + [ + "ĠF", + "r" + ], + [ + "Ġbad", + "ly" + ], + [ + "Ġhair", + "y" + ], + [ + "Ġcra", + "wl" + ], + [ + "Ġlive", + "ly" + ], + [ + "Ġstep", + "s" + ], + [ + "ÅĵLet", + "â" + ], + [ + "Ġbrace", + "let" + ], + [ + "Ġcere", + "al" + ], + [ + "Ġthirst", + "y" + ], + [ + "Ġcarp", + "et" + ], + [ + "Ġw", + "ool" + ], + [ + "it", + "her" + ], + [ + "Ġgo", + "al" + ], + [ + "Ġback", + "pack" + ], + [ + "Ġtr", + "uth" + ], + [ + "Ġcom", + "pl" + ], + [ + "Ġche", + "ap" + ], + [ + "Ġdis", + "g" + ], + [ + "Ġplac", + "ed" + ], + [ + "Ġear", + "ly" + ], + [ + "Ġdecor", + "ate" + ], + [ + "Ġnut", + "s" + ], + [ + "O", + "w" + ], + [ + "c", + "ream" + ], + [ + "Ġb", + "ump" + ], + [ + "Ġthem", + "selves" + ], + [ + "ic", + "es" + ], + [ + "Ġhelp", + "less" + ], + [ + "Ġcl", + "um" + ], + [ + "Ġsc", + "atter" + ], + [ + "Ġsc", + "ale" + ], + [ + "Ġnew", + "s" + ], + [ + "ust", + "ing" + ], + [ + "Ġen", + "v" + ], + [ + "âĤ¬â", + "Ģ" + ], + [ + "Ġtell", + "ing" + ], + [ + "Ġswing", + "ing" + ], + [ + "Ġspark", + "led" + ], + [ + "Ġcarry", + "ing" + ], + [ + "gg", + "ing" + ], + [ + "in", + "es" + ], + [ + "ri", + "c" + ], + [ + "Ġfriends", + "hip" + ], + [ + "Ġcare", + "less" + ], + [ + "Ġsn", + "e" + ], + [ + "Ġdoes", + "n" + ], + [ + "Ġfall", + "ing" + ], + [ + "Ġcomp", + "ut" + ], + [ + "Ġpa", + "le" + ], + [ + "Ġpee", + "ked" + ], + [ + "Ġcompassion", + "ate" + ], + [ + "c", + "om" + ], + [ + "v", + "ice" + ], + [ + "Ġs", + "end" + ], + [ + "Ġst", + "age" + ], + [ + "Ġme", + "m" + ], + [ + "Ġch", + "alk" + ], + [ + "Ġch", + "asing" + ], + [ + "Ġpe", + "bble" + ], + [ + "ĠEvery", + "thing" + ], + [ + "Ġcu", + "be" + ], + [ + "Ġpig", + "e" + ], + [ + "Ġcou", + "rage" + ], + [ + "Ġfing", + "ers" + ], + [ + "Ġturt", + "le" + ], + [ + "b", + "led" + ], + [ + "Ġs", + "ink" + ], + [ + "Ġb", + "icycle" + ], + [ + "Ġsh", + "aking" + ], + [ + "Ġsleep", + "ing" + ], + [ + "Ġneighb", + "our" + ], + [ + "Ġmoder", + "n" + ], + [ + "Ġdisg", + "usting" + ], + [ + "Ġclum", + "sy" + ], + [ + "g", + "est" + ], + [ + "Ġh", + "ook" + ], + [ + "Ġc", + "her" + ], + [ + "Ġn", + "ail" + ], + [ + "Ġbig", + "gest" + ], + [ + "ic", + "hes" + ], + [ + "Ġbu", + "l" + ], + [ + "Ġend", + "ing" + ], + [ + "Ġde", + "ad" + ], + [ + "Ġtri", + "ang" + ], + [ + "ul", + "ance" + ], + [ + "Ġspo", + "ke" + ], + [ + "Ġpract", + "iced" + ], + [ + "Ġamb", + "ulance" + ], + [ + "Ġcomput", + "er" + ], + [ + "i", + "bb" + ], + [ + "Ġl", + "amp" + ], + [ + "Ġst", + "ation" + ], + [ + "Ġch", + "ar" + ], + [ + "Ġwall", + "et" + ], + [ + "ĠTo", + "day" + ], + [ + "Ġpack", + "age" + ], + [ + "Ġmic", + "rop" + ], + [ + "Ġcush", + "ion" + ], + [ + "keep", + "er" + ], + [ + "Ġcrocod", + "ile" + ], + [ + "Ġmicrop", + "hone" + ], + [ + "u", + "nder" + ], + [ + "ĠA", + "lex" + ], + [ + "Ġthought", + "ful" + ], + [ + "Ġjo", + "lly" + ], + [ + "Ġpen", + "gu" + ], + [ + "Ġpump", + "kin" + ], + [ + "Ġcost", + "um" + ], + [ + "Ġf", + "ive" + ], + [ + "Ġd", + "ough" + ], + [ + "Ġp", + "ony" + ], + [ + "Ġo", + "l" + ], + [ + "im", + "i" + ], + [ + "Ġst", + "airs" + ], + [ + "Ġkn", + "ock" + ], + [ + "Ġgra", + "nd" + ], + [ + "Ġint", + "ell" + ], + [ + "Ġun", + "iver" + ], + [ + "Ġbright", + "er" + ], + [ + "Ġopen", + "s" + ], + [ + "Ġdream", + "ing" + ], + [ + "Ġboun", + "ced" + ], + [ + "Ġquest", + "ion" + ], + [ + "Ġstat", + "ue" + ], + [ + "Ġw", + "ine" + ], + [ + "is", + "k" + ], + [ + "ĠA", + "lice" + ], + [ + "Ġco", + "coa" + ], + [ + "ĠYou", + "r" + ], + [ + "Ġpe", + "pper" + ], + [ + "Ġbeaut", + "y" + ], + [ + "Ġper", + "m" + ], + [ + "Ġpain", + "ting" + ], + [ + "Ġsho", + "e" + ], + [ + "Ġele", + "v" + ], + [ + "Every", + "one" + ], + [ + "hin", + "o" + ], + [ + "Ġweal", + "thy" + ], + [ + "Ġintell", + "ig" + ], + [ + "n", + "oon" + ], + [ + "Ġs", + "uit" + ], + [ + "Ġm", + "ild" + ], + [ + "Ġn", + "ature" + ], + [ + "an", + "s" + ], + [ + "Ġhapp", + "ier" + ], + [ + "Ġne", + "at" + ], + [ + "Ġstart", + "s" + ], + [ + "uck", + "ed" + ], + [ + "Ġafter", + "noon" + ], + [ + "Ġtor", + "n" + ], + [ + "Ġelev", + "ator" + ], + [ + "g", + "en" + ], + [ + "t", + "i" + ], + [ + "Ġt", + "en" + ], + [ + "Ġh", + "arsh" + ], + [ + "Ġp", + "ed" + ], + [ + "at", + "ient" + ], + [ + "or", + "ant" + ], + [ + "Ġre", + "ce" + ], + [ + "Ġj", + "et" + ], + [ + "ill", + "ie" + ], + [ + "ĠR", + "ed" + ], + [ + "Ġread", + "ing" + ], + [ + "Ġsear", + "ching" + ], + [ + "Ġbasket", + "ball" + ], + [ + "Ġbar", + "ber" + ], + [ + "Ġspe", + "ed" + ], + [ + "S", + "ee" + ], + [ + "a", + "wn" + ], + [ + "Ġst", + "ack" + ], + [ + "ĠM", + "ummy" + ], + [ + "Ġclo", + "ck" + ], + [ + "ĠG", + "ive" + ], + [ + "Ġmag", + "n" + ], + [ + "Ġlea", + "ving" + ], + [ + "Ġcup", + "board" + ], + [ + "Ġdes", + "ign" + ], + [ + "Ġslid", + "es" + ], + [ + "Ġsandw", + "iches" + ], + [ + "Ġcand", + "le" + ], + [ + "aul", + "if" + ], + [ + "Ġrid", + "ing" + ], + [ + "Ġfre", + "sh" + ], + [ + "o", + "ppy" + ], + [ + "Ġsh", + "ocked" + ], + [ + "Ġen", + "er" + ], + [ + "Ġjump", + "s" + ], + [ + "Ġhair", + "cut" + ], + [ + "Ġrespect", + "ful" + ], + [ + "Ġcoo", + "king" + ], + [ + "Ġtick", + "et" + ], + [ + "Ġenc", + "ou" + ], + [ + "Ġtunn", + "el" + ], + [ + "aulif", + "lower" + ], + [ + "G", + "ive" + ], + [ + "Ġs", + "le" + ], + [ + "re", + "en" + ], + [ + "Ġm", + "ay" + ], + [ + "Ġth", + "ief" + ], + [ + "Ġy", + "awn" + ], + [ + "Ġfor", + "t" + ], + [ + "Ġal", + "ert" + ], + [ + "Ġsor", + "ts" + ], + [ + "Ġimp", + "atient" + ], + [ + "Ġcre", + "ate" + ], + [ + "ail", + "able" + ], + [ + "Ġwhe", + "els" + ], + [ + "Ġval", + "ue" + ], + [ + "Ġpri", + "ze" + ], + [ + "M", + "ary" + ], + [ + "y", + "e" + ], + [ + "he", + "t" + ], + [ + "Ġs", + "ent" + ], + [ + "Ġb", + "ent" + ], + [ + "in", + "ce" + ], + [ + "Ġc", + "omet" + ], + [ + "Ġc", + "amp" + ], + [ + "Ġc", + "auliflower" + ], + [ + "Ġm", + "elon" + ], + [ + "Ġg", + "em" + ], + [ + "Ġit", + "self" + ], + [ + "ir", + "on" + ], + [ + "Ġmu", + "sh" + ], + [ + "Ġmonkey", + "s" + ], + [ + "Ġsal", + "ad" + ], + [ + "Ġdest", + "ro" + ], + [ + "Ġresc", + "ue" + ], + [ + "Ġscoot", + "er" + ], + [ + "Ġintellig", + "ent" + ], + [ + "a", + "z" + ], + [ + "Ġd", + "ug" + ], + [ + "il", + "ing" + ], + [ + "an", + "ger" + ], + [ + "Ġr", + "oof" + ], + [ + "Ġr", + "hino" + ], + [ + "Ġsc", + "old" + ], + [ + "ag", + "het" + ], + [ + "Ġexp", + "er" + ], + [ + "Ġbark", + "ing" + ], + [ + "Ġbanan", + "as" + ], + [ + "Ġign", + "orant" + ], + [ + "Ġmic", + "ro" + ], + [ + "Ġav", + "ailable" + ], + [ + "Ġgrace", + "ful" + ], + [ + "Ġmotorcy", + "cle" + ], + [ + "aghet", + "ti" + ], + [ + "h", + "ood" + ], + [ + "ĠS", + "n" + ], + [ + "Ġe", + "arth" + ], + [ + "Ġbo", + "ots" + ], + [ + "Ġsp", + "aghetti" + ], + [ + "um", + "mer" + ], + [ + "Ġag", + "ree" + ], + [ + "Ġbu", + "ll" + ], + [ + "Ġany", + "where" + ], + [ + "ĠG", + "eorge" + ], + [ + "Ġbeh", + "ave" + ], + [ + "ĠK", + "im" + ], + [ + "Ġpa", + "id" + ], + [ + "sc", + "ope" + ], + [ + "Ġstret", + "ched" + ], + [ + "Ġfear", + "ful" + ], + [ + "Ġav", + "o" + ], + [ + "Ġmicro", + "scope" + ], + [ + "I", + "f" + ], + [ + "i", + "o" + ], + [ + "Ġnot", + "eb" + ], + [ + "ĠE", + "ventually" + ], + [ + "Ġold", + "er" + ], + [ + "Ġsn", + "iff" + ], + [ + "Ġad", + "vice" + ], + [ + "Ġstop", + "s" + ], + [ + "Ġperf", + "orm" + ], + [ + "Ġfur", + "ther" + ], + [ + "Ġenv", + "ious" + ], + [ + "Ġpige", + "on" + ], + [ + "Ġmush", + "room" + ], + [ + "Ġnoteb", + "ook" + ], + [ + "u", + "bby" + ], + [ + "Ġp", + "un" + ], + [ + "at", + "s" + ], + [ + "Ġn", + "urse" + ], + [ + "or", + "able" + ], + [ + "Ġth", + "under" + ], + [ + "Ġcl", + "apping" + ], + [ + "Ġv", + "ide" + ], + [ + "Ġbu", + "ilt" + ], + [ + "ĠC", + "l" + ], + [ + "Ġun", + "com" + ], + [ + "Ġshout", + "ing" + ], + [ + "Ġar", + "row" + ], + [ + "Ġdraw", + "er" + ], + [ + "Ġpro", + "v" + ], + [ + "fort", + "able" + ], + [ + "Ġtom", + "ato" + ], + [ + "Ġmeet", + "ing" + ], + [ + "Ġob", + "ject" + ], + [ + "Ġyog", + "urt" + ], + [ + "Ġtemp", + "le" + ], + [ + "Ġuncom", + "fortable" + ], + [ + "e", + "f" + ], + [ + "u", + "es" + ], + [ + "ĠT", + "ony" + ], + [ + "Ġloo", + "p" + ], + [ + "Ġch", + "ubby" + ], + [ + "Ġsw", + "itch" + ], + [ + "Ġsk", + "ip" + ], + [ + "Ġad", + "orable" + ], + [ + "Åĵ", + "It" + ], + [ + "Ġcount", + "ing" + ], + [ + "Ġcoo", + "ked" + ], + [ + "Ġp", + "ist" + ], + [ + "ar", + "ian" + ], + [ + "en", + "ry" + ], + [ + "Ġst", + "ared" + ], + [ + "Ġsh", + "ut" + ], + [ + "ic", + "op" + ], + [ + "Ġqu", + "ite" + ], + [ + "Ġsn", + "uggled" + ], + [ + "ĠGra", + "ndpa" + ], + [ + "Ġtrou", + "bled" + ], + [ + "Ġgigg", + "le" + ], + [ + "Ġblow", + "ing" + ], + [ + "Ġhel", + "icop" + ], + [ + "Ġfis", + "her" + ], + [ + "Ġpist", + "ol" + ], + [ + "Ġhelicop", + "ter" + ], + [ + "p", + "ar" + ], + [ + "ĠS", + "ophie" + ], + [ + "oo", + "ped" + ], + [ + "ip", + "s" + ], + [ + "Ġnow", + "here" + ], + [ + "Ġforg", + "ave" + ], + [ + "Ġmar", + "ched" + ], + [ + "Ġtreasure", + "s" + ], + [ + "ho", + "st" + ], + [ + "Ġpass", + "port" + ], + [ + "Ġstir", + "red" + ], + [ + "Ġpus", + "hing" + ], + [ + "Ġcompetit", + "ive" + ], + [ + "Ġdestro", + "y" + ], + [ + "Y", + "ay" + ], + [ + "a", + "ked" + ], + [ + "o", + "e" + ], + [ + "t", + "ter" + ], + [ + "Ġg", + "ir" + ], + [ + "Ġg", + "host" + ], + [ + "an", + "ic" + ], + [ + "ke", + "let" + ], + [ + "ht", + "ub" + ], + [ + "Ġre", + "ef" + ], + [ + "Ġse", + "par" + ], + [ + "op", + "ard" + ], + [ + "pl", + "ay" + ], + [ + "Ġen", + "vel" + ], + [ + "Ġbre", + "at" + ], + [ + "Ġrep", + "air" + ], + [ + "Ġfoot", + "ball" + ], + [ + "Ġbat", + "htub" + ], + [ + "Ġrubb", + "er" + ], + [ + "Ġang", + "el" + ], + [ + "Ġtriang", + "le" + ], + [ + "Ġl", + "ed" + ], + [ + "Ġm", + "ill" + ], + [ + "ĠS", + "h" + ], + [ + "ch", + "anic" + ], + [ + "Ġnam", + "es" + ], + [ + "Ġle", + "opard" + ], + [ + "Ġno", + "body" + ], + [ + "Ġany", + "way" + ], + [ + "uc", + "t" + ], + [ + "ct", + "us" + ], + [ + "Ġshould", + "n" + ], + [ + "Ġz", + "eb" + ], + [ + "Ġgi", + "ven" + ], + [ + "Ġhold", + "s" + ], + [ + "Ġbar", + "rel" + ], + [ + "Ġband", + "age" + ], + [ + "Ġent", + "h" + ], + [ + "Dad", + "dy" + ], + [ + "Ġzeb", + "ra" + ], + [ + "i", + "ves" + ], + [ + "u", + "be" + ], + [ + "y", + "ear" + ], + [ + "Ġp", + "or" + ], + [ + "Ġp", + "and" + ], + [ + "Ġo", + "tter" + ], + [ + "Ġr", + "ibb" + ], + [ + "Ġsh", + "ield" + ], + [ + "Ġme", + "chanic" + ], + [ + "Ġdon", + "â" + ], + [ + "Ġdis", + "ag" + ], + [ + "Ġdis", + "play" + ], + [ + "Ġmusic", + "ian" + ], + [ + "cer", + "os" + ], + [ + "Ġspe", + "ak" + ], + [ + "Ġca", + "b" + ], + [ + "Ġca", + "ctus" + ], + [ + "ĠBet", + "sy" + ], + [ + "Ġbul", + "b" + ], + [ + "C", + "h" + ], + [ + "e", + "ath" + ], + [ + "i", + "kes" + ], + [ + "x", + "y" + ], + [ + "Ġt", + "ube" + ], + [ + "Ġa", + "head" + ], + [ + "Ġs", + "ummer" + ], + [ + "Ġs", + "kelet" + ], + [ + "Ġp", + "urse" + ], + [ + "Ġkn", + "ob" + ], + [ + "Ġdidn", + "â" + ], + [ + "Ġcouldn", + "â" + ], + [ + "Ġcol", + "ours" + ], + [ + "Ġwindow", + "s" + ], + [ + "oom", + "y" + ], + [ + "Ġgif", + "ted" + ], + [ + "Ġcart", + "oon" + ], + [ + "Ġrhino", + "ceros" + ], + [ + "g", + "l" + ], + [ + "he", + "art" + ], + [ + "Ġs", + "ize" + ], + [ + "Ġb", + "et" + ], + [ + "ĠM", + "iss" + ], + [ + "Ġsc", + "rew" + ], + [ + "Ġfl", + "our" + ], + [ + "Ġbl", + "in" + ], + [ + "ap", + "a" + ], + [ + "Ġdis", + "hes" + ], + [ + "Ġrain", + "ing" + ], + [ + "Ġhop", + "ing" + ], + [ + "Ġsho", + "vel" + ], + [ + "Ġmin", + "t" + ], + [ + "Ġjelly", + "fish" + ], + [ + "Ġswe", + "ater" + ], + [ + "Ġchar", + "ming" + ], + [ + "Ġpengu", + "in" + ], + [ + "Ġenvel", + "ope" + ], + [ + "Ġenth", + "us" + ], + [ + "Ġcab", + "in" + ], + [ + "re", + "c" + ], + [ + "Ġp", + "ant" + ], + [ + "Ġth", + "read" + ], + [ + "Ġin", + "se" + ], + [ + "Ġst", + "aff" + ], + [ + "Ġli", + "z" + ], + [ + "Ġun", + "pack" + ], + [ + "Ġwr", + "iting" + ], + [ + "Ġtra", + "sh" + ], + [ + "Ġroll", + "ing" + ], + [ + "Ġwaff", + "le" + ], + [ + "Ġ", + "iron" + ], + [ + "Ġw", + "ing" + ], + [ + "er", + "able" + ], + [ + "Ġshe", + "et" + ], + [ + "ĠB", + "uddy" + ], + [ + "Ġsh", + "adow" + ], + [ + "Ġro", + "ared" + ], + [ + "Ġmu", + "ff" + ], + [ + "Ġair", + "port" + ], + [ + "Ġce", + "iling" + ], + [ + "Ġatt", + "ic" + ], + [ + "Ġorgan", + "ize" + ], + [ + "Ġstret", + "ch" + ], + [ + "Ġgrap", + "es" + ], + [ + "Ġo", + "bs" + ], + [ + "ĠA", + "re" + ], + [ + "ĠW", + "ould" + ], + [ + "ĠIn", + "st" + ], + [ + "Ġpi", + "ano" + ], + [ + "Ġsal", + "t" + ], + [ + "Ġmis", + "erable" + ], + [ + "Ġc", + "ord" + ], + [ + "Ġd", + "ess" + ], + [ + "Ġre", + "fused" + ], + [ + "Ġsc", + "reen" + ], + [ + "ĠD", + "an" + ], + [ + "Ġstr", + "ugg" + ], + [ + "Ġneed", + "le" + ], + [ + "Ġbear", + "s" + ], + [ + "ĠAnd", + "y" + ], + [ + "Ġhead", + "ed" + ], + [ + "Ġstick", + "y" + ], + [ + "Ġfree", + "z" + ], + [ + "Ġzoom", + "ed" + ], + [ + "Ġimag", + "ined" + ], + [ + "Ġmed", + "al" + ], + [ + "Ġmagn", + "et" + ], + [ + "rec", + "i" + ], + [ + "Ġobs", + "er" + ], + [ + "Ġ", + "er" + ], + [ + "Ġli", + "ves" + ], + [ + "ĠA", + "l" + ], + [ + "ĠP", + "at" + ], + [ + "Ġapp", + "reci" + ], + [ + "ass", + "es" + ], + [ + "Ġsail", + "or" + ], + [ + "Ġminut", + "e" + ], + [ + "Ġfisher", + "man" + ], + [ + "v", + "ision" + ], + [ + "Ġr", + "at" + ], + [ + "al", + "ed" + ], + [ + "Ġbr", + "us" + ], + [ + "Ġch", + "ance" + ], + [ + "Ġpo", + "st" + ], + [ + "Ġsun", + "gl" + ], + [ + "Ġche", + "ek" + ], + [ + "ree", + "ze" + ], + [ + "Ġde", + "af" + ], + [ + "Ġru", + "les" + ], + [ + "Ġfrog", + "s" + ], + [ + "Ġair", + "pl" + ], + [ + "ĠGo", + "d" + ], + [ + "Ġstrawber", + "ry" + ], + [ + "Ġsle", + "pt" + ], + [ + "Ġliz", + "ard" + ], + [ + "Ġsungl", + "asses" + ], + [ + "Ġl", + "uck" + ], + [ + "Ġin", + "f" + ], + [ + "Ġbe", + "ep" + ], + [ + "ĠB", + "unny" + ], + [ + "Ġme", + "as" + ], + [ + "Ġro", + "d" + ], + [ + "Ġor", + "der" + ], + [ + "Ġrest", + "less" + ], + [ + "Ġsand", + "box" + ], + [ + "Ġmess", + "age" + ], + [ + "ĠJoe", + "y" + ], + [ + "Ġcomp", + "let" + ], + [ + "Ġatt", + "ract" + ], + [ + "Ġgold", + "en" + ], + [ + "Gra", + "ndma" + ], + [ + "Ġpil", + "ot" + ], + [ + "Ġpor", + "ch" + ], + [ + "L", + "ittle" + ], + [ + "l", + "er" + ], + [ + "s", + "es" + ] + ] + } +} \ No newline at end of file diff --git a/outio/mlp-linear-9L_run/tokenizer_config.json b/outio/mlp-linear-9L_run/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..f1578d3b6947e13e4001d559b79d4944be52b9c1 --- /dev/null +++ b/outio/mlp-linear-9L_run/tokenizer_config.json @@ -0,0 +1,13 @@ +{ + "add_prefix_space": false, + "backend": "tokenizers", + "bos_token": "<|endoftext|>", + "eos_token": "<|endoftext|>", + "errors": "replace", + "is_local": false, + "local_files_only": false, + "model_max_length": 1024, + "pad_token": "<|endoftext|>", + "tokenizer_class": "GPT2Tokenizer", + "unk_token": "<|endoftext|>" +} diff --git a/outio/mlp-linear-9L_run/training_args.bin b/outio/mlp-linear-9L_run/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..b3fbb54ac1ee27b6a4fde07a6a6bd316742ee7a5 --- /dev/null +++ b/outio/mlp-linear-9L_run/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fa9985a7c7f521fa2a6f7d16c7d254b6d5e243707cd393fa6ea6dfe9b9926147 +size 4920 diff --git a/outio/mlp-linear-9L_run/training_log.jsonl b/outio/mlp-linear-9L_run/training_log.jsonl new file mode 100644 index 0000000000000000000000000000000000000000..1aa827dccf93e5cb9ebbd4c4f17354cd5a0ea426 --- /dev/null +++ b/outio/mlp-linear-9L_run/training_log.jsonl @@ -0,0 +1,54 @@ +{"step": 20, "epoch": 0.026963262554769128, "timestamp": 1786254867.8820844, "loss": 120.13565673828126, "grad_norm": 19.875, "learning_rate": 0.0005, "train/total_time_seconds": 23.281483590602875, "train/time_per_step_avg": 1.1640741795301437, "train/epoch_time_elapsed": 42.8496764190495, "train/estimated_remaining_minutes": 14.162902517616748, "train/global/act/norm": 45865.29424772523, "train/global/act/mean": 0.0008468838877243953, "train/global/act/std": 0.405950734162207, "train/global/act/max_abs": 8.328570365905762, "train/global/act/frac_near_dtype_limit": 0.0, "train/global/act/frac_near_user_limit": 0.0, "train/global/grad/norm": 7.223068074765932, "train/global/grad/mean": 1.5862884322552294e-06, "train/global/grad/std": 0.0012770101063532878, "train/global/grad/max_abs": 0.1865234375, "train/global/grad/frac_near_dtype_limit": 0.0, "train/global/grad/frac_near_user_limit": 0.0, "train/global/param/norm": 56.8370559383082, "train/global/param/mean": 0.0012002339254830034, "train/global/param/std": 0.040175776880214266, "train/global/param/max_abs": 1.0, "train/global/param/frac_near_dtype_limit": 0.0, "train/global/param/frac_near_user_limit": 0.0, "train/layer__model_layers_4/param/norm": 17.9352881367118, "train/layer__model_layers_4/param/mean": 0.0015472673960669364, "train/layer__model_layers_4/param/std": 0.044252915450776684, "train/layer__model_layers_4/param/max_abs": 1.0, "train/layer__model_layers_4/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_4/param/frac_near_user_limit": 0.0, "train/layer__model_layers_7/param/norm": 17.939752417379538, "train/layer__model_layers_7/param/mean": 0.001527842790957732, "train/layer__model_layers_7/param/std": 0.04426481997403276, "train/layer__model_layers_7/param/max_abs": 1.0, "train/layer__model_layers_7/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_7/param/frac_near_user_limit": 0.0, "train/layer__model_layers_6/param/norm": 17.92636985918022, "train/layer__model_layers_6/param/mean": 0.001631773950157225, "train/layer__model_layers_6/param/std": 0.04422802810460986, "train/layer__model_layers_6/param/max_abs": 1.0, "train/layer__model_layers_6/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_6/param/frac_near_user_limit": 0.0, "train/layer_model_layers_6/act/norm": 14130.943375884945, "train/layer_model_layers_6/act/mean": 0.001090640058884254, "train/layer_model_layers_6/act/std": 0.42790409106686467, "train/layer_model_layers_6/act/max_abs": 5.125, "train/layer_model_layers_6/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_6/act/frac_near_user_limit": 0.0, "train/layer_model_layers_6/grad/norm": 1.1282536394103617, "train/layer_model_layers_6/grad/mean": -7.801037481528316e-07, "train/layer_model_layers_6/grad/std": 0.0006963687337265827, "train/layer_model_layers_6/grad/max_abs": 0.007659912109375, "train/layer_model_layers_6/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_6/grad/frac_near_user_limit": 0.0, "train/layer_model_layers_7/act/norm": 14162.553496833301, "train/layer_model_layers_7/act/mean": 0.0004149681100478539, "train/layer_model_layers_7/act/std": 0.42889934741429514, "train/layer_model_layers_7/act/max_abs": 5.5, "train/layer_model_layers_7/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_7/act/frac_near_user_limit": 0.0, "train/layer_model_layers_7/grad/norm": 1.0656749934054095, "train/layer_model_layers_7/grad/mean": -4.2460115153165227e-07, "train/layer_model_layers_7/grad/std": 0.000657913163571747, "train/layer_model_layers_7/grad/max_abs": 0.00848388671875, "train/layer_model_layers_7/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_7/grad/frac_near_user_limit": 0.0, "train/layer__model_layers_8/param/norm": 17.918659803138876, "train/layer__model_layers_8/param/mean": 0.0014265063400387577, "train/layer__model_layers_8/param/std": 0.044202242984042614, "train/layer__model_layers_8/param/max_abs": 1.0, "train/layer__model_layers_8/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_8/param/frac_near_user_limit": 0.0, "train/layer_model_layers_4/act/norm": 14079.863892348538, "train/layer_model_layers_4/act/mean": 0.0009709768570386446, "train/layer_model_layers_4/act/std": 0.4264060735955697, "train/layer_model_layers_4/act/max_abs": 5.25, "train/layer_model_layers_4/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_4/act/frac_near_user_limit": 0.0, "train/layer_model_layers_4/grad/norm": 1.4315023400178932, "train/layer_model_layers_4/grad/mean": -2.028000673247202e-06, "train/layer_model_layers_4/grad/std": 0.0008836196933127202, "train/layer_model_layers_4/grad/max_abs": 0.00927734375, "train/layer_model_layers_4/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_4/grad/frac_near_user_limit": 0.0, "train/layer_model_layers_8/act/norm": 14188.43172167708, "train/layer_model_layers_8/act/mean": 0.0013178988144947933, "train/layer_model_layers_8/act/std": 0.4296572584517151, "train/layer_model_layers_8/act/max_abs": 5.5625, "train/layer_model_layers_8/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_8/act/frac_near_user_limit": 0.0, "train/layer_model_layers_8/grad/norm": 1.0206292167726159, "train/layer_model_layers_8/grad/mean": 5.3110055200640815e-08, "train/layer_model_layers_8/grad/std": 0.000630075835746595, "train/layer_model_layers_8/grad/max_abs": 0.0072021484375, "train/layer_model_layers_8/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_8/grad/frac_near_user_limit": 0.0, "train/layer_model_layers_5/act/norm": 14117.1315554156, "train/layer_model_layers_5/act/mean": 0.0018905794847971545, "train/layer_model_layers_5/act/std": 0.4275287071610389, "train/layer_model_layers_5/act/max_abs": 5.28125, "train/layer_model_layers_5/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_5/act/frac_near_user_limit": 0.0, "train/layer_model_layers_5/grad/norm": 1.2122966690033181, "train/layer_model_layers_5/grad/mean": -9.892893230103759e-07, "train/layer_model_layers_5/grad/std": 0.0007485524005112892, "train/layer_model_layers_5/grad/max_abs": 0.00830078125, "train/layer_model_layers_5/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_5/grad/frac_near_user_limit": 0.0, "train/layer__model_layers_1/param/norm": 17.933035158611606, "train/layer__model_layers_1/param/mean": 0.0015046377077861436, "train/layer__model_layers_1/param/std": 0.04424885735283964, "train/layer__model_layers_1/param/max_abs": 1.0, "train/layer__model_layers_1/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_1/param/frac_near_user_limit": 0.0, "train/layer_model_layers_3/act/norm": 14066.883930544302, "train/layer_model_layers_3/act/mean": 0.0025618718220637394, "train/layer_model_layers_3/act/std": 0.42613855008964696, "train/layer_model_layers_3/act/max_abs": 4.84375, "train/layer_model_layers_3/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_3/act/frac_near_user_limit": 0.0, "train/layer_model_layers_3/grad/norm": 1.7070210406409134, "train/layer_model_layers_3/grad/mean": 1.9531665070566482e-06, "train/layer_model_layers_3/grad/std": 0.001054004791488333, "train/layer_model_layers_3/grad/max_abs": 0.0103759765625, "train/layer_model_layers_3/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_3/grad/frac_near_user_limit": 0.0, "train/layer_model_layers_1/act/norm": 14001.340936971721, "train/layer_model_layers_1/act/mean": 7.794453547551215e-07, "train/layer_model_layers_1/act/std": 0.4240053066896342, "train/layer_model_layers_1/act/max_abs": 5.3125, "train/layer_model_layers_1/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_1/act/frac_near_user_limit": 0.0, "train/layer_model_layers_1/grad/norm": 2.6490285904002473, "train/layer_model_layers_1/grad/mean": 2.259144367399159e-06, "train/layer_model_layers_1/grad/std": 0.001634984444366829, "train/layer_model_layers_1/grad/max_abs": 0.0179443359375, "train/layer_model_layers_1/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_1/grad/frac_near_user_limit": 0.0, "train/layer__model_layers_3/param/norm": 17.937513610622013, "train/layer__model_layers_3/param/mean": 0.0014466227681700042, "train/layer__model_layers_3/param/std": 0.044262013477257604, "train/layer__model_layers_3/param/max_abs": 1.0, "train/layer__model_layers_3/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_3/param/frac_near_user_limit": 0.0, "train/layer__model_layers_2/param/norm": 17.933035158611606, "train/layer__model_layers_2/param/mean": 0.0015101395605506837, "train/layer__model_layers_2/param/std": 0.044265292316262486, "train/layer__model_layers_2/param/max_abs": 1.0, "train/layer__model_layers_2/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_2/param/frac_near_user_limit": 0.0, "train/layer__model_layers_5/param/norm": 17.929821963003427, "train/layer__model_layers_5/param/mean": 0.001554492111325078, "train/layer__model_layers_5/param/std": 0.04425848149488436, "train/layer__model_layers_5/param/max_abs": 1.0, "train/layer__model_layers_5/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_5/param/frac_near_user_limit": 0.0, "train/layer_model_layers_0/act/norm": 13969.97237396381, "train/layer_model_layers_0/act/mean": -0.0001725346709673221, "train/layer_model_layers_0/act/std": 0.42352815051836784, "train/layer_model_layers_0/act/max_abs": 4.75, "train/layer_model_layers_0/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_0/act/frac_near_user_limit": 0.0, "train/layer_model_layers_0/grad/norm": 4.140204517701514, "train/layer_model_layers_0/grad/mean": -4.7352227277751553e-07, "train/layer_model_layers_0/grad/std": 0.002555505245858976, "train/layer_model_layers_0/grad/max_abs": 0.0361328125, "train/layer_model_layers_0/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_0/grad/frac_near_user_limit": 0.0, "train/layer__model_layers_0/param/norm": 17.93527452440093, "train/layer__model_layers_0/param/mean": 0.001573654865139918, "train/layer__model_layers_0/param/std": 0.044252048307771956, "train/layer__model_layers_0/param/max_abs": 1.0, "train/layer__model_layers_0/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_0/param/frac_near_user_limit": 0.0, "train/layer_model_layers_2/act/norm": 14034.215279860322, "train/layer_model_layers_2/act/mean": 0.0025705924400916467, "train/layer_model_layers_2/act/std": 0.4250360320197809, "train/layer_model_layers_2/act/max_abs": 4.90625, "train/layer_model_layers_2/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_2/act/frac_near_user_limit": 0.0, "train/layer_model_layers_2/grad/norm": 2.2317679743784002, "train/layer_model_layers_2/grad/mean": 2.4813947300702718e-06, "train/layer_model_layers_2/grad/std": 0.0013771787384382957, "train/layer_model_layers_2/grad/max_abs": 0.0146484375, "train/layer_model_layers_2/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_2/grad/frac_near_user_limit": 0.0, "train/tensor_act_model_embed_tokens/norm": 183.14187215122737, "train/tensor_act_model_embed_tokens/mean": -3.91155481338501e-05, "train/tensor_act_model_embed_tokens/std": 0.02001953329041244, "train/tensor_act_model_embed_tokens/max_abs": 0.09619140625, "train/tensor_act_model_embed_tokens/frac_near_dtype_limit": 0.0, "train/tensor_act_model_embed_tokens/frac_near_user_limit": 0.0, "train/tensor_act_model_rotary_emb/norm": 3632.2067871093745, "train/tensor_act_model_rotary_emb/mean": 0.333984375, "train/tensor_act_model_rotary_emb/std": 0.71875, "train/tensor_act_model_rotary_emb/max_abs": 1.0, "train/tensor_act_model_rotary_emb/frac_near_dtype_limit": 0.0, "train/tensor_act_model_rotary_emb/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_input_layernorm/norm": 9146.768310589385, "train/tensor_act_model_layers_0_input_layernorm/mean": -0.0023627281188964844, "train/tensor_act_model_layers_0_input_layernorm/std": 1.000000091723332, "train/tensor_act_model_layers_0_input_layernorm/max_abs": 4.375, "train/tensor_act_model_layers_0_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_q_proj/norm": 2073.7606174205857, "train/tensor_act_model_layers_0_self_attn_q_proj/mean": -0.0006031990051269532, "train/tensor_act_model_layers_0_self_attn_q_proj/std": 0.2265625557654413, "train/tensor_act_model_layers_0_self_attn_q_proj/max_abs": 1.0703125, "train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_k_proj/norm": 2089.1339244416013, "train/tensor_act_model_layers_0_self_attn_k_proj/mean": 8.234567940235139e-05, "train/tensor_act_model_layers_0_self_attn_k_proj/std": 0.2281499629770971, "train/tensor_act_model_layers_0_self_attn_k_proj/max_abs": 1.0234375, "train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_v_proj/norm": 2107.624164923062, "train/tensor_act_model_layers_0_self_attn_v_proj/mean": -0.0013871192932128906, "train/tensor_act_model_layers_0_self_attn_v_proj/std": 0.23046878774569188, "train/tensor_act_model_layers_0_self_attn_v_proj/max_abs": 1.0703125, "train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_o_proj/norm": 78.02399889508851, "train/tensor_act_model_layers_0_self_attn_o_proj/mean": 0.0001543760299682617, "train/tensor_act_model_layers_0_self_attn_o_proj/std": 0.008522998259386573, "train/tensor_act_model_layers_0_self_attn_o_proj/max_abs": 0.2578125, "train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn/norm": 78.02399889508851, "train/tensor_act_model_layers_0_self_attn/mean": 0.0001543760299682617, "train/tensor_act_model_layers_0_self_attn/std": 0.008522998259386573, "train/tensor_act_model_layers_0_self_attn/max_abs": 0.2578125, "train/tensor_act_model_layers_0_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_post_attention_layernorm/norm": 9148.850097786815, "train/tensor_act_model_layers_0_post_attention_layernorm/mean": 0.004792213439941406, "train/tensor_act_model_layers_0_post_attention_layernorm/std": 1.000002007208682, "train/tensor_act_model_layers_0_post_attention_layernorm/max_abs": 4.75, "train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_mlp_up_proj/norm": 3560.757550565122, "train/tensor_act_model_layers_0_mlp_up_proj/mean": -7.581710815429688e-05, "train/tensor_act_model_layers_0_mlp_up_proj/std": 0.2246094857487788, "train/tensor_act_model_layers_0_mlp_up_proj/max_abs": 1.1796875, "train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_mlp_down_proj/norm": 808.1675831830216, "train/tensor_act_model_layers_0_mlp_down_proj/mean": -0.000986814498901367, "train/tensor_act_model_layers_0_mlp_down_proj/std": 0.08834877783806921, "train/tensor_act_model_layers_0_mlp_down_proj/max_abs": 0.51171875, "train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_mlp/norm": 808.1675831830216, "train/tensor_act_model_layers_0_mlp/mean": -0.000986814498901367, "train/tensor_act_model_layers_0_mlp/std": 0.08834877783806921, "train/tensor_act_model_layers_0_mlp/max_abs": 0.51171875, "train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0/norm": 831.1538419700142, "train/tensor_act_model_layers_0/mean": -0.0008721351623535156, "train/tensor_act_model_layers_0/std": 0.09082059435858165, "train/tensor_act_model_layers_0/max_abs": 0.52734375, "train/tensor_act_model_layers_0/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_input_layernorm/norm": 9158.358703626907, "train/tensor_act_model_layers_1_input_layernorm/mean": -0.008217811584472656, "train/tensor_act_model_layers_1_input_layernorm/std": 1.00000304894128, "train/tensor_act_model_layers_1_input_layernorm/max_abs": 5.3125, "train/tensor_act_model_layers_1_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_q_proj/norm": 2034.8610171837768, "train/tensor_act_model_layers_1_self_attn_q_proj/mean": -0.0059032440185546875, "train/tensor_act_model_layers_1_self_attn_q_proj/std": 0.2220469414917578, "train/tensor_act_model_layers_1_self_attn_q_proj/max_abs": 1.1875, "train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_k_proj/norm": 2082.27089692581, "train/tensor_act_model_layers_1_self_attn_k_proj/mean": 0.00048154592514038086, "train/tensor_act_model_layers_1_self_attn_k_proj/std": 0.22747849840050277, "train/tensor_act_model_layers_1_self_attn_k_proj/max_abs": 1.171875, "train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_v_proj/norm": 2050.4426360629855, "train/tensor_act_model_layers_1_self_attn_v_proj/mean": -0.000854790210723877, "train/tensor_act_model_layers_1_self_attn_v_proj/std": 0.22369433159830898, "train/tensor_act_model_layers_1_self_attn_v_proj/max_abs": 1.1875, "train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_o_proj/norm": 180.58027982936176, "train/tensor_act_model_layers_1_self_attn_o_proj/mean": 0.0011816024780273438, "train/tensor_act_model_layers_1_self_attn_o_proj/std": 0.01968680419952855, "train/tensor_act_model_layers_1_self_attn_o_proj/max_abs": 0.2421875, "train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn/norm": 180.58027982936176, "train/tensor_act_model_layers_1_self_attn/mean": 0.0011816024780273438, "train/tensor_act_model_layers_1_self_attn/std": 0.01968680419952855, "train/tensor_act_model_layers_1_self_attn/max_abs": 0.2421875, "train/tensor_act_model_layers_1_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_post_attention_layernorm/norm": 9158.379089361999, "train/tensor_act_model_layers_1_post_attention_layernorm/mean": 0.004539966583251953, "train/tensor_act_model_layers_1_post_attention_layernorm/std": 1.0000031951497133, "train/tensor_act_model_layers_1_post_attention_layernorm/max_abs": 5.25, "train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_mlp_up_proj/norm": 3579.287503193819, "train/tensor_act_model_layers_1_mlp_up_proj/mean": 0.0012593269348144531, "train/tensor_act_model_layers_1_mlp_up_proj/std": 0.2255861107294601, "train/tensor_act_model_layers_1_mlp_up_proj/max_abs": 1.1875, "train/tensor_act_model_layers_1_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_mlp_down_proj/norm": 821.0661844308066, "train/tensor_act_model_layers_1_mlp_down_proj/mean": 0.0011713504791259768, "train/tensor_act_model_layers_1_mlp_down_proj/std": 0.08963061949140268, "train/tensor_act_model_layers_1_mlp_down_proj/max_abs": 0.4765625, "train/tensor_act_model_layers_1_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_mlp/norm": 821.0661844308066, "train/tensor_act_model_layers_1_mlp/mean": 0.0011713504791259768, "train/tensor_act_model_layers_1_mlp/std": 0.08963061949140268, "train/tensor_act_model_layers_1_mlp/max_abs": 0.4765625, "train/tensor_act_model_layers_1_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1/norm": 1174.9310743433161, "train/tensor_act_model_layers_1/mean": 0.0014805793762207031, "train/tensor_act_model_layers_1/std": 0.12805229517561986, "train/tensor_act_model_layers_1/max_abs": 0.640625, "train/tensor_act_model_layers_1/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_input_layernorm/norm": 9158.63647461776, "train/tensor_act_model_layers_2_input_layernorm/mean": 0.011058807373046875, "train/tensor_act_model_layers_2_input_layernorm/std": 1.0000010433645015, "train/tensor_act_model_layers_2_input_layernorm/max_abs": 4.875, "train/tensor_act_model_layers_2_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_q_proj/norm": 2065.775366237947, "train/tensor_act_model_layers_2_self_attn_q_proj/mean": -0.011943817138671875, "train/tensor_act_model_layers_2_self_attn_q_proj/std": 0.22522063394043246, "train/tensor_act_model_layers_2_self_attn_q_proj/max_abs": 1.3828125, "train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_k_proj/norm": 2092.924444260376, "train/tensor_act_model_layers_2_self_attn_k_proj/mean": 0.008089065551757809, "train/tensor_act_model_layers_2_self_attn_k_proj/std": 0.2284549606805223, "train/tensor_act_model_layers_2_self_attn_k_proj/max_abs": 1.296875, "train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_v_proj/norm": 2096.376890747997, "train/tensor_act_model_layers_2_self_attn_v_proj/mean": -0.008035659790039062, "train/tensor_act_model_layers_2_self_attn_v_proj/std": 0.22869949776392487, "train/tensor_act_model_layers_2_self_attn_v_proj/max_abs": 1.2421875, "train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_o_proj/norm": 226.16535075261754, "train/tensor_act_model_layers_2_self_attn_o_proj/mean": 0.0012426376342773438, "train/tensor_act_model_layers_2_self_attn_o_proj/std": 0.024668824166795517, "train/tensor_act_model_layers_2_self_attn_o_proj/max_abs": 0.228515625, "train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn/norm": 226.16535075261754, "train/tensor_act_model_layers_2_self_attn/mean": 0.0012426376342773438, "train/tensor_act_model_layers_2_self_attn/std": 0.024668824166795517, "train/tensor_act_model_layers_2_self_attn/max_abs": 0.228515625, "train/tensor_act_model_layers_2_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_post_attention_layernorm/norm": 9158.648437517744, "train/tensor_act_model_layers_2_post_attention_layernorm/mean": 0.0203094482421875, "train/tensor_act_model_layers_2_post_attention_layernorm/std": 1.000001718172259, "train/tensor_act_model_layers_2_post_attention_layernorm/max_abs": 4.90625, "train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_mlp_up_proj/norm": 3570.9513244909417, "train/tensor_act_model_layers_2_mlp_up_proj/mean": 0.0018334388732910156, "train/tensor_act_model_layers_2_mlp_up_proj/std": 0.22509841710844916, "train/tensor_act_model_layers_2_mlp_up_proj/max_abs": 1.28125, "train/tensor_act_model_layers_2_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_mlp_down_proj/norm": 790.1283816142175, "train/tensor_act_model_layers_2_mlp_down_proj/mean": 0.0010769367218017578, "train/tensor_act_model_layers_2_mlp_down_proj/std": 0.08633508027126516, "train/tensor_act_model_layers_2_mlp_down_proj/max_abs": 0.47265625, "train/tensor_act_model_layers_2_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_mlp/norm": 790.1283816142175, "train/tensor_act_model_layers_2_mlp/mean": 0.0010769367218017578, "train/tensor_act_model_layers_2_mlp/std": 0.08633508027126516, "train/tensor_act_model_layers_2_mlp/max_abs": 0.47265625, "train/tensor_act_model_layers_2_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2/norm": 1432.682090175319, "train/tensor_act_model_layers_2/mean": 0.0038003921508789062, "train/tensor_act_model_layers_2/std": 0.15625055417960487, "train/tensor_act_model_layers_2/max_abs": 0.8359375, "train/tensor_act_model_layers_2/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_input_layernorm/norm": 9158.735351567297, "train/tensor_act_model_layers_3_input_layernorm/mean": 0.02446746826171875, "train/tensor_act_model_layers_3_input_layernorm/std": 1.0000034318886946, "train/tensor_act_model_layers_3_input_layernorm/max_abs": 4.78125, "train/tensor_act_model_layers_3_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_q_proj/norm": 2092.894250965192, "train/tensor_act_model_layers_3_self_attn_q_proj/mean": 0.002222776412963867, "train/tensor_act_model_layers_3_self_attn_q_proj/std": 0.22851672499565404, "train/tensor_act_model_layers_3_self_attn_q_proj/max_abs": 1.1953125, "train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_k_proj/norm": 2052.9258628307634, "train/tensor_act_model_layers_3_self_attn_k_proj/mean": -1.609325408935547e-05, "train/tensor_act_model_layers_3_self_attn_k_proj/std": 0.22418327748137298, "train/tensor_act_model_layers_3_self_attn_k_proj/max_abs": 1.1328125, "train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_v_proj/norm": 2073.2135136194793, "train/tensor_act_model_layers_3_self_attn_v_proj/mean": -0.006084442138671875, "train/tensor_act_model_layers_3_self_attn_v_proj/std": 0.2263803761978546, "train/tensor_act_model_layers_3_self_attn_v_proj/max_abs": 1.203125, "train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_o_proj/norm": 223.4814746474767, "train/tensor_act_model_layers_3_self_attn_o_proj/mean": 0.0013904571533203125, "train/tensor_act_model_layers_3_self_attn_o_proj/std": 0.02435658282244344, "train/tensor_act_model_layers_3_self_attn_o_proj/max_abs": 0.2255859375, "train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn/norm": 223.4814746474767, "train/tensor_act_model_layers_3_self_attn/mean": 0.0013904571533203125, "train/tensor_act_model_layers_3_self_attn/std": 0.02435658282244344, "train/tensor_act_model_layers_3_self_attn/max_abs": 0.2255859375, "train/tensor_act_model_layers_3_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_post_attention_layernorm/norm": 9158.734741215712, "train/tensor_act_model_layers_3_post_attention_layernorm/mean": 0.03290557861328125, "train/tensor_act_model_layers_3_post_attention_layernorm/std": 1.0000052582043284, "train/tensor_act_model_layers_3_post_attention_layernorm/max_abs": 4.84375, "train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_mlp_up_proj/norm": 3594.7566943629945, "train/tensor_act_model_layers_3_mlp_up_proj/mean": -0.0059356689453125, "train/tensor_act_model_layers_3_mlp_up_proj/std": 0.22656261638317735, "train/tensor_act_model_layers_3_mlp_up_proj/max_abs": 1.296875, "train/tensor_act_model_layers_3_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_mlp_down_proj/norm": 829.2458262629668, "train/tensor_act_model_layers_3_mlp_down_proj/mean": -0.0034513473510742188, "train/tensor_act_model_layers_3_mlp_down_proj/std": 0.09039351059760549, "train/tensor_act_model_layers_3_mlp_down_proj/max_abs": 0.431640625, "train/tensor_act_model_layers_3_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_mlp/norm": 829.2458262629668, "train/tensor_act_model_layers_3_mlp/mean": -0.0034513473510742188, "train/tensor_act_model_layers_3_mlp/std": 0.09039351059760549, "train/tensor_act_model_layers_3_mlp/max_abs": 0.431640625, "train/tensor_act_model_layers_3_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3/norm": 1679.8756164997528, "train/tensor_act_model_layers_3/mean": 0.001737833023071289, "train/tensor_act_model_layers_3/std": 0.18353348927121566, "train/tensor_act_model_layers_3/max_abs": 0.984375, "train/tensor_act_model_layers_3/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_input_layernorm/norm": 9158.782958988662, "train/tensor_act_model_layers_4_input_layernorm/mean": 0.009237289428710938, "train/tensor_act_model_layers_4_input_layernorm/std": 1.0000033612375674, "train/tensor_act_model_layers_4_input_layernorm/max_abs": 5.0625, "train/tensor_act_model_layers_4_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_q_proj/norm": 2066.455070534267, "train/tensor_act_model_layers_4_self_attn_q_proj/mean": -0.009563446044921875, "train/tensor_act_model_layers_4_self_attn_q_proj/std": 0.22540440079163515, "train/tensor_act_model_layers_4_self_attn_q_proj/max_abs": 1.203125, "train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_k_proj/norm": 2100.324965169152, "train/tensor_act_model_layers_4_self_attn_k_proj/mean": 0.0060024261474609375, "train/tensor_act_model_layers_4_self_attn_k_proj/std": 0.22943195828941673, "train/tensor_act_model_layers_4_self_attn_k_proj/max_abs": 1.1875, "train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_v_proj/norm": 2030.455616326556, "train/tensor_act_model_layers_4_self_attn_v_proj/mean": -0.004791259765624999, "train/tensor_act_model_layers_4_self_attn_v_proj/std": 0.2216801315147726, "train/tensor_act_model_layers_4_self_attn_v_proj/max_abs": 1.171875, "train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_o_proj/norm": 250.93902724467202, "train/tensor_act_model_layers_4_self_attn_o_proj/mean": 0.0006207823753356934, "train/tensor_act_model_layers_4_self_attn_o_proj/std": 0.027385711249535913, "train/tensor_act_model_layers_4_self_attn_o_proj/max_abs": 0.2373046875, "train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn/norm": 250.93902724467202, "train/tensor_act_model_layers_4_self_attn/mean": 0.0006207823753356934, "train/tensor_act_model_layers_4_self_attn/std": 0.027385711249535913, "train/tensor_act_model_layers_4_self_attn/max_abs": 0.2373046875, "train/tensor_act_model_layers_4_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_post_attention_layernorm/norm": 9158.788757328228, "train/tensor_act_model_layers_4_post_attention_layernorm/mean": 0.012622833251953125, "train/tensor_act_model_layers_4_post_attention_layernorm/std": 1.0000021418577287, "train/tensor_act_model_layers_4_post_attention_layernorm/max_abs": 5.25, "train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_mlp_up_proj/norm": 3555.8362572354035, "train/tensor_act_model_layers_4_mlp_up_proj/mean": -0.00014309585094451904, "train/tensor_act_model_layers_4_mlp_up_proj/std": 0.22418291445047742, "train/tensor_act_model_layers_4_mlp_up_proj/max_abs": 1.234375, "train/tensor_act_model_layers_4_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_mlp_down_proj/norm": 811.0240760131593, "train/tensor_act_model_layers_4_mlp_down_proj/mean": -0.0013518333435058594, "train/tensor_act_model_layers_4_mlp_down_proj/std": 0.08850153981303953, "train/tensor_act_model_layers_4_mlp_down_proj/max_abs": 0.484375, "train/tensor_act_model_layers_4_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_mlp/norm": 811.0240760131593, "train/tensor_act_model_layers_4_mlp/mean": -0.0013518333435058594, "train/tensor_act_model_layers_4_mlp/std": 0.08850153981303953, "train/tensor_act_model_layers_4_mlp/max_abs": 0.484375, "train/tensor_act_model_layers_4_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4/norm": 1893.6912433949042, "train/tensor_act_model_layers_4/mean": 0.0010062456130981445, "train/tensor_act_model_layers_4/std": 0.20678798842065674, "train/tensor_act_model_layers_4/max_abs": 1.140625, "train/tensor_act_model_layers_4/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_input_layernorm/norm": 9158.812072765204, "train/tensor_act_model_layers_5_input_layernorm/mean": 0.004483461380004883, "train/tensor_act_model_layers_5_input_layernorm/std": 1.0000021717879695, "train/tensor_act_model_layers_5_input_layernorm/max_abs": 5.25, "train/tensor_act_model_layers_5_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_q_proj/norm": 2100.6405818001044, "train/tensor_act_model_layers_5_self_attn_q_proj/mean": 0.01197052001953125, "train/tensor_act_model_layers_5_self_attn_q_proj/std": 0.22906618031553744, "train/tensor_act_model_layers_5_self_attn_q_proj/max_abs": 1.1953125, "train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_k_proj/norm": 2051.5121024181535, "train/tensor_act_model_layers_5_self_attn_k_proj/mean": 0.001157522201538086, "train/tensor_act_model_layers_5_self_attn_k_proj/std": 0.22393874789383927, "train/tensor_act_model_layers_5_self_attn_k_proj/max_abs": 1.1640625, "train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_v_proj/norm": 2075.85469408804, "train/tensor_act_model_layers_5_self_attn_v_proj/mean": 0.00022197276121005416, "train/tensor_act_model_layers_5_self_attn_v_proj/std": 0.22662457978015543, "train/tensor_act_model_layers_5_self_attn_v_proj/max_abs": 1.2578125, "train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_o_proj/norm": 254.5151423734983, "train/tensor_act_model_layers_5_self_attn_o_proj/mean": 0.0008132457733154297, "train/tensor_act_model_layers_5_self_attn_o_proj/std": 0.02778320703029389, "train/tensor_act_model_layers_5_self_attn_o_proj/max_abs": 0.220703125, "train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn/norm": 254.5151423734983, "train/tensor_act_model_layers_5_self_attn/mean": 0.0008132457733154297, "train/tensor_act_model_layers_5_self_attn/std": 0.02778320703029389, "train/tensor_act_model_layers_5_self_attn/max_abs": 0.220703125, "train/tensor_act_model_layers_5_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_post_attention_layernorm/norm": 9158.816345220177, "train/tensor_act_model_layers_5_post_attention_layernorm/mean": 0.008457183837890625, "train/tensor_act_model_layers_5_post_attention_layernorm/std": 1.0000021982608298, "train/tensor_act_model_layers_5_post_attention_layernorm/max_abs": 5.28125, "train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_mlp_up_proj/norm": 3573.0466421568854, "train/tensor_act_model_layers_5_mlp_up_proj/mean": -0.0011423826217651367, "train/tensor_act_model_layers_5_mlp_up_proj/std": 0.22540349692495276, "train/tensor_act_model_layers_5_mlp_up_proj/max_abs": 1.1875, "train/tensor_act_model_layers_5_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_mlp_down_proj/norm": 799.3515006834078, "train/tensor_act_model_layers_5_mlp_down_proj/mean": -0.0005774497985839844, "train/tensor_act_model_layers_5_mlp_down_proj/std": 0.08740284699963051, "train/tensor_act_model_layers_5_mlp_down_proj/max_abs": 0.462890625, "train/tensor_act_model_layers_5_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_mlp/norm": 799.3515006834078, "train/tensor_act_model_layers_5_mlp/mean": -0.0005774497985839844, "train/tensor_act_model_layers_5_mlp/std": 0.08740284699963051, "train/tensor_act_model_layers_5_mlp/max_abs": 0.462890625, "train/tensor_act_model_layers_5_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5/norm": 2102.6028122912185, "train/tensor_act_model_layers_5/mean": 0.0012424290180206299, "train/tensor_act_model_layers_5/std": 0.22961531855631992, "train/tensor_act_model_layers_5/max_abs": 1.2421875, "train/tensor_act_model_layers_5/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_input_layernorm/norm": 9158.833007819172, "train/tensor_act_model_layers_6_input_layernorm/mean": 0.005686476826667785, "train/tensor_act_model_layers_6_input_layernorm/std": 1.0000025146593547, "train/tensor_act_model_layers_6_input_layernorm/max_abs": 5.125, "train/tensor_act_model_layers_6_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_q_proj/norm": 2024.70933516154, "train/tensor_act_model_layers_6_self_attn_q_proj/mean": 0.008039474487304688, "train/tensor_act_model_layers_6_self_attn_q_proj/std": 0.22082672623882046, "train/tensor_act_model_layers_6_self_attn_q_proj/max_abs": 1.1875, "train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_k_proj/norm": 2052.910089178976, "train/tensor_act_model_layers_6_self_attn_k_proj/mean": -0.01146697998046875, "train/tensor_act_model_layers_6_self_attn_k_proj/std": 0.22381880042885957, "train/tensor_act_model_layers_6_self_attn_k_proj/max_abs": 1.1171875, "train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_v_proj/norm": 2090.9327299066867, "train/tensor_act_model_layers_6_self_attn_v_proj/mean": 0.01435089111328125, "train/tensor_act_model_layers_6_self_attn_v_proj/std": 0.22772393664304802, "train/tensor_act_model_layers_6_self_attn_v_proj/max_abs": 1.265625, "train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_o_proj/norm": 248.66062033676886, "train/tensor_act_model_layers_6_self_attn_o_proj/mean": 0.0006154179573059082, "train/tensor_act_model_layers_6_self_attn_o_proj/std": 0.027142998303483507, "train/tensor_act_model_layers_6_self_attn_o_proj/max_abs": 0.2236328125, "train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn/norm": 248.66062033676886, "train/tensor_act_model_layers_6_self_attn/mean": 0.0006154179573059082, "train/tensor_act_model_layers_6_self_attn/std": 0.027142998303483507, "train/tensor_act_model_layers_6_self_attn/max_abs": 0.2236328125, "train/tensor_act_model_layers_6_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_post_attention_layernorm/norm": 9158.83203125751, "train/tensor_act_model_layers_6_post_attention_layernorm/mean": 0.008283138275146484, "train/tensor_act_model_layers_6_post_attention_layernorm/std": 1.0000041493962357, "train/tensor_act_model_layers_6_post_attention_layernorm/max_abs": 5.09375, "train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_mlp_up_proj/norm": 3565.7061851279122, "train/tensor_act_model_layers_6_mlp_up_proj/mean": -0.004791259765625, "train/tensor_act_model_layers_6_mlp_up_proj/std": 0.2246097779788673, "train/tensor_act_model_layers_6_mlp_up_proj/max_abs": 1.2265625, "train/tensor_act_model_layers_6_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_mlp_down_proj/norm": 821.2329713025825, "train/tensor_act_model_layers_6_mlp_down_proj/mean": 0.0001903623342514038, "train/tensor_act_model_layers_6_mlp_down_proj/std": 0.08963114505276222, "train/tensor_act_model_layers_6_mlp_down_proj/max_abs": 0.52734375, "train/tensor_act_model_layers_6_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_mlp/norm": 821.2329713025825, "train/tensor_act_model_layers_6_mlp/mean": 0.0001903623342514038, "train/tensor_act_model_layers_6_mlp/std": 0.08963114505276222, "train/tensor_act_model_layers_6_mlp/max_abs": 0.52734375, "train/tensor_act_model_layers_6_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6/norm": 2245.545601906354, "train/tensor_act_model_layers_6/mean": 0.0020475387573242188, "train/tensor_act_model_layers_6/std": 0.24524060481554763, "train/tensor_act_model_layers_6/max_abs": 1.515625, "train/tensor_act_model_layers_6/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_input_layernorm/norm": 9158.841613782926, "train/tensor_act_model_layers_7_input_layernorm/mean": 0.0080718994140625, "train/tensor_act_model_layers_7_input_layernorm/std": 1.0000036152712841, "train/tensor_act_model_layers_7_input_layernorm/max_abs": 5.5, "train/tensor_act_model_layers_7_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_q_proj/norm": 2094.243451175893, "train/tensor_act_model_layers_7_self_attn_q_proj/mean": -0.0008530020713806152, "train/tensor_act_model_layers_7_self_attn_q_proj/std": 0.2287007431041426, "train/tensor_act_model_layers_7_self_attn_q_proj/max_abs": 1.15625, "train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_k_proj/norm": 2070.246830624731, "train/tensor_act_model_layers_7_self_attn_k_proj/mean": -0.0004820823669433594, "train/tensor_act_model_layers_7_self_attn_k_proj/std": 0.22607542128992095, "train/tensor_act_model_layers_7_self_attn_k_proj/max_abs": 1.21875, "train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_v_proj/norm": 2034.9890479320418, "train/tensor_act_model_layers_7_self_attn_v_proj/mean": -0.005054473876953125, "train/tensor_act_model_layers_7_self_attn_v_proj/std": 0.22210886029447952, "train/tensor_act_model_layers_7_self_attn_v_proj/max_abs": 1.1953125, "train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_o_proj/norm": 244.24156031417377, "train/tensor_act_model_layers_7_self_attn_o_proj/mean": -0.0015883445739746094, "train/tensor_act_model_layers_7_self_attn_o_proj/std": 0.026614338566017588, "train/tensor_act_model_layers_7_self_attn_o_proj/max_abs": 0.228515625, "train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn/norm": 244.24156031417377, "train/tensor_act_model_layers_7_self_attn/mean": -0.0015883445739746094, "train/tensor_act_model_layers_7_self_attn/std": 0.026614338566017588, "train/tensor_act_model_layers_7_self_attn/max_abs": 0.228515625, "train/tensor_act_model_layers_7_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_post_attention_layernorm/norm": 9158.846984883812, "train/tensor_act_model_layers_7_post_attention_layernorm/mean": 0.0016190111637115479, "train/tensor_act_model_layers_7_post_attention_layernorm/std": 1.0000036257512501, "train/tensor_act_model_layers_7_post_attention_layernorm/max_abs": 5.40625, "train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_mlp_up_proj/norm": 3593.3060081563526, "train/tensor_act_model_layers_7_mlp_up_proj/mean": -0.00042695552110671997, "train/tensor_act_model_layers_7_mlp_up_proj/std": 0.22656268376185978, "train/tensor_act_model_layers_7_mlp_up_proj/max_abs": 1.2734375, "train/tensor_act_model_layers_7_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_mlp_down_proj/norm": 790.095908420183, "train/tensor_act_model_layers_7_mlp_down_proj/mean": 0.0020303726196289062, "train/tensor_act_model_layers_7_mlp_down_proj/std": 0.08627386209580751, "train/tensor_act_model_layers_7_mlp_down_proj/max_abs": 0.458984375, "train/tensor_act_model_layers_7_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_mlp/norm": 790.095908420183, "train/tensor_act_model_layers_7_mlp/mean": 0.0020303726196289062, "train/tensor_act_model_layers_7_mlp/std": 0.08627386209580751, "train/tensor_act_model_layers_7_mlp/max_abs": 0.458984375, "train/tensor_act_model_layers_7_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7/norm": 2390.9091821424804, "train/tensor_act_model_layers_7/mean": 0.0024900436401367188, "train/tensor_act_model_layers_7/std": 0.2612325815294042, "train/tensor_act_model_layers_7/max_abs": 1.5859375, "train/tensor_act_model_layers_7/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_input_layernorm/norm": 9158.859985368514, "train/tensor_act_model_layers_8_input_layernorm/mean": 0.009693145751953127, "train/tensor_act_model_layers_8_input_layernorm/std": 1.00000284127924, "train/tensor_act_model_layers_8_input_layernorm/max_abs": 5.5625, "train/tensor_act_model_layers_8_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_q_proj/norm": 2055.8883424244605, "train/tensor_act_model_layers_8_self_attn_q_proj/mean": -0.0019369125366210938, "train/tensor_act_model_layers_8_self_attn_q_proj/std": 0.22442729413672624, "train/tensor_act_model_layers_8_self_attn_q_proj/max_abs": 1.1484375, "train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_k_proj/norm": 2068.8739646072754, "train/tensor_act_model_layers_8_self_attn_k_proj/mean": -0.0036063194274902344, "train/tensor_act_model_layers_8_self_attn_k_proj/std": 0.22570981265453868, "train/tensor_act_model_layers_8_self_attn_k_proj/max_abs": 1.3203125, "train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_v_proj/norm": 2068.861311438254, "train/tensor_act_model_layers_8_self_attn_v_proj/mean": 0.008327484130859375, "train/tensor_act_model_layers_8_self_attn_v_proj/std": 0.22570992749941532, "train/tensor_act_model_layers_8_self_attn_v_proj/max_abs": 1.234375, "train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_o_proj/norm": 252.43754597084012, "train/tensor_act_model_layers_8_self_attn_o_proj/mean": 0.0007788538932800293, "train/tensor_act_model_layers_8_self_attn_o_proj/std": 0.02753775469306995, "train/tensor_act_model_layers_8_self_attn_o_proj/max_abs": 0.2490234375, "train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn/norm": 252.43754597084012, "train/tensor_act_model_layers_8_self_attn/mean": 0.0007788538932800293, "train/tensor_act_model_layers_8_self_attn/std": 0.02753775469306995, "train/tensor_act_model_layers_8_self_attn/max_abs": 0.2490234375, "train/tensor_act_model_layers_8_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_post_attention_layernorm/norm": 9158.848205577773, "train/tensor_act_model_layers_8_post_attention_layernorm/mean": 0.012630462646484375, "train/tensor_act_model_layers_8_post_attention_layernorm/std": 1.0000028783574142, "train/tensor_act_model_layers_8_post_attention_layernorm/max_abs": 5.5, "train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_mlp_up_proj/norm": 3580.865034948006, "train/tensor_act_model_layers_8_mlp_up_proj/mean": 0.00017113983631134033, "train/tensor_act_model_layers_8_mlp_up_proj/std": 0.2256475356138171, "train/tensor_act_model_layers_8_mlp_up_proj/max_abs": 1.3671875, "train/tensor_act_model_layers_8_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_mlp_down_proj/norm": 809.8662183360632, "train/tensor_act_model_layers_8_mlp_down_proj/mean": -0.0044384002685546875, "train/tensor_act_model_layers_8_mlp_down_proj/std": 0.08825753633051737, "train/tensor_act_model_layers_8_mlp_down_proj/max_abs": 0.515625, "train/tensor_act_model_layers_8_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_mlp/norm": 809.8662183360632, "train/tensor_act_model_layers_8_mlp/mean": -0.0044384002685546875, "train/tensor_act_model_layers_8_mlp/std": 0.08825753633051737, "train/tensor_act_model_layers_8_mlp/max_abs": 0.515625, "train/tensor_act_model_layers_8_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8/norm": 2548.227647097366, "train/tensor_act_model_layers_8/mean": -0.0011695027351379395, "train/tensor_act_model_layers_8/std": 0.27820098038329805, "train/tensor_act_model_layers_8/max_abs": 1.6171875, "train/tensor_act_model_layers_8/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8/frac_near_user_limit": 0.0, "train/tensor_act_model_norm/norm": 9158.86169434579, "train/tensor_act_model_norm/mean": -0.004282474517822266, "train/tensor_act_model_norm/std": 1.0000037655109555, "train/tensor_act_model_norm/max_abs": 5.40625, "train/tensor_act_model_norm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_norm/frac_near_user_limit": 0.0, "train/tensor_act_model/norm": 9158.86169434579, "train/tensor_act_model/mean": -0.004282474517822266, "train/tensor_act_model/std": 1.0000037655109555, "train/tensor_act_model/max_abs": 5.40625, "train/tensor_act_model/frac_near_dtype_limit": 0.0, "train/tensor_act_model/frac_near_user_limit": 0.0, "train/tensor_act_lm_head/norm": 11725.399748464048, "train/tensor_act_lm_head/mean": -0.002635955810546875, "train/tensor_act_lm_head/std": 0.22656260601017586, "train/tensor_act_lm_head/max_abs": 1.3046875, "train/tensor_act_lm_head/frac_near_dtype_limit": 0.0, "train/tensor_act_lm_head/frac_near_user_limit": 0.0, "train/tensor_act_/norm": 33.30561320829714, "train/tensor_act_/mean": 8.326403200626372, "train/tensor_act_/std": 0.0, "train/tensor_act_/max_abs": 8.328570365905762, "train/tensor_act_/frac_near_dtype_limit": 0.0, "train/tensor_act_/frac_near_user_limit": 0.0, "train/tensor_grad_model_norm_weight/norm": 0.04188363882885196, "train/tensor_grad_model_norm_weight/mean": 0.0002678632736206055, "train/tensor_grad_model_norm_weight/std": 0.0008890040181361811, "train/tensor_grad_model_norm_weight/max_abs": 0.0028839111328125, "train/tensor_grad_model_norm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_norm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/norm": 0.6567039736043584, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/mean": -9.6810981631279e-07, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/std": 0.0007410135862090632, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/max_abs": 0.006195068359375, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/norm": 0.6515465164968496, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/mean": 8.295464795082808e-07, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/std": 0.000734766420680706, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/max_abs": 0.0072021484375, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/norm": 0.01359782430549884, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/mean": -2.0615756511688232e-05, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/std": 0.0003009088541924834, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/max_abs": 0.0018768310546875, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/norm": 0.2942602555165827, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/mean": 1.780688762664795e-06, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/std": 0.0005747273290534046, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/max_abs": 0.0050048828125, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/norm": 0.31477405922147494, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/mean": -6.610644049942493e-07, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/std": 0.0006147932948693195, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/max_abs": 0.006011962890625, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/norm": 0.001750753845104515, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/mean": -2.6364432414993644e-08, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/std": 3.419446456839792e-06, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/max_abs": 4.00543212890625e-05, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/norm": 0.0017998297966950977, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/mean": 1.907392288558185e-08, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/std": 3.5152944340062413e-06, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/max_abs": 3.361701965332031e-05, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_input_layernorm_weight/norm": 0.0069292377166829105, "train/tensor_grad_model_layers_8_input_layernorm_weight/mean": -4.675639502238482e-07, "train/tensor_grad_model_layers_8_input_layernorm_weight/std": 0.00015373373291674794, "train/tensor_grad_model_layers_8_input_layernorm_weight/max_abs": 0.000705718994140625, "train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/norm": 0.7141149589250648, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/mean": -2.352462615817785e-07, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/std": 0.0008054169332475143, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/max_abs": 0.006134033203125, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/norm": 0.6564795682913401, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/mean": -3.329805622342974e-07, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/std": 0.0007407871686613992, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/max_abs": 0.00848388671875, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/norm": 0.012920870122618318, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/mean": 3.403838491067289e-07, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/std": 0.00028670502751474385, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/max_abs": 0.00121307373046875, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/norm": 0.3028268980337274, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/mean": -1.289532519876957e-06, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/std": 0.0005914590815408416, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/max_abs": 0.004180908203125, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/norm": 0.3206335180389203, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/mean": -1.4007091522216797e-06, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/std": 0.0006262375418262379, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/max_abs": 0.004974365234375, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/norm": 0.0021741666189685196, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/mean": -3.8708094507455826e-08, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/std": 4.246421515060917e-06, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/max_abs": 5.364418029785156e-05, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/norm": 0.002567798513375777, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/mean": -3.0499904823955144e-09, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/std": 5.0152389834807915e-06, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/max_abs": 6.437301635742188e-05, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_input_layernorm_weight/norm": 0.007024306935671228, "train/tensor_grad_model_layers_7_input_layernorm_weight/mean": 2.321600914001465e-05, "train/tensor_grad_model_layers_7_input_layernorm_weight/std": 0.00015395906571011544, "train/tensor_grad_model_layers_7_input_layernorm_weight/max_abs": 0.000885009765625, "train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/norm": 0.7438265720988673, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/mean": -2.0747538655996327e-06, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/std": 0.0008384931686117645, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/max_abs": 0.005950927734375, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/norm": 0.6993919456413047, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/mean": -3.5176053643226624e-06, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/std": 0.0007891246411385598, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/max_abs": 0.007659912109375, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/norm": 0.01472038244348852, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/mean": 8.637085556983948e-06, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/std": 0.0003264389268301327, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/max_abs": 0.00183868408203125, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/norm": 0.34701616318845746, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/mean": 7.051974534988403e-06, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/std": 0.0006777664703541301, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/max_abs": 0.005035400390625, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/norm": 0.33140774695504827, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/mean": 1.7903803382068872e-06, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/std": 0.0006472813368124866, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/max_abs": 0.005767822265625, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/norm": 0.0022034954129965608, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/mean": -1.4113084034761412e-08, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/std": 4.303716816398566e-06, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/max_abs": 6.198883056640625e-05, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/norm": 0.0021091067565236657, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/mean": -6.679329089820386e-08, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/std": 4.119352436859102e-06, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/max_abs": 3.933906555175781e-05, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_input_layernorm_weight/norm": 0.007204596862996378, "train/tensor_grad_model_layers_6_input_layernorm_weight/mean": 1.7270445823669434e-05, "train/tensor_grad_model_layers_6_input_layernorm_weight/std": 0.0001589390006203737, "train/tensor_grad_model_layers_6_input_layernorm_weight/max_abs": 0.0006866455078125, "train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/norm": 0.747758695306347, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/mean": -3.4459662856534123e-07, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/std": 0.000843754609081815, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/max_abs": 0.0072021484375, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/norm": 0.7760631037744542, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/mean": -2.996530383825302e-06, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/std": 0.0008756717372626727, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/max_abs": 0.00830078125, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/norm": 0.017407784045094903, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/mean": 3.597140312194824e-05, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/std": 0.0003842433628006087, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/max_abs": 0.00136566162109375, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/norm": 0.41318894853685667, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/mean": 2.4502514861524105e-07, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/std": 0.0008070099821470191, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/max_abs": 0.005706787109375, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/norm": 0.37033420933695094, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/mean": -3.3291871659457684e-07, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/std": 0.0007233092280854832, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/max_abs": 0.007568359375, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/norm": 0.002440637228965039, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/mean": -3.9814040064811707e-08, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/std": 4.766879032479843e-06, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/max_abs": 6.389617919921875e-05, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/norm": 0.002526216840896826, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/mean": -3.939203452318907e-08, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/std": 4.934022192967669e-06, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/max_abs": 5.054473876953125e-05, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_input_layernorm_weight/norm": 0.0076356404487144145, "train/tensor_grad_model_layers_5_input_layernorm_weight/mean": 1.4121178537607193e-07, "train/tensor_grad_model_layers_5_input_layernorm_weight/std": 0.00016932095066547807, "train/tensor_grad_model_layers_5_input_layernorm_weight/max_abs": 0.000720977783203125, "train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/norm": 0.882807638779481, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/mean": 5.54340658709407e-07, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/std": 0.0009956603127519643, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/max_abs": 0.0081787109375, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/norm": 0.9121088522147858, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/mean": -5.8319419622421265e-06, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/std": 0.0010288251542380133, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/max_abs": 0.006866455078125, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/norm": 0.01801550628411251, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/mean": 2.5529414415359497e-05, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/std": 0.0003992425801541201, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/max_abs": 0.00174713134765625, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/norm": 0.46096025830998805, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/mean": -1.1819647625088692e-06, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/std": 0.0009003131973670483, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/max_abs": 0.00927734375, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/norm": 0.474326971987658, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/mean": -3.4016557037830353e-06, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/std": 0.000926420855110026, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/max_abs": 0.006683349609375, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/norm": 0.002494692435009133, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/mean": 6.381014827638865e-08, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/std": 4.872465469938096e-06, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/max_abs": 4.935264587402344e-05, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/norm": 0.0026511816667665676, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/mean": -2.6222551241517067e-08, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/std": 5.178094080596203e-06, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/max_abs": 4.744529724121094e-05, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_input_layernorm_weight/norm": 0.009940688270318328, "train/tensor_grad_model_layers_4_input_layernorm_weight/mean": -1.693516969680786e-05, "train/tensor_grad_model_layers_4_input_layernorm_weight/std": 0.00021997906412640544, "train/tensor_grad_model_layers_4_input_layernorm_weight/max_abs": 0.00101470947265625, "train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/norm": 1.0388073141763723, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/mean": 1.942069502547383e-06, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/std": 0.0011718623522265713, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/max_abs": 0.0103759765625, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/norm": 1.0325180118643056, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/mean": 2.0922016119584437e-06, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/std": 0.00116522222764922, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/max_abs": 0.0091552734375, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/norm": 0.020544972132187877, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/mean": -7.319450378417969e-05, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/std": 0.0004498627859734642, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/max_abs": 0.00141143798828125, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/norm": 0.6386494763847563, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/mean": 5.392357707023621e-06, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/std": 0.0012478367145385219, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/max_abs": 0.008544921875, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/norm": 0.6001999521867529, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/mean": 2.564862370491028e-06, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/std": 0.001172265692604121, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/max_abs": 0.01031494140625, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/norm": 0.0040055756316137005, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/mean": 1.5011755749583244e-07, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/std": 7.821619938549773e-06, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/max_abs": 0.00010395050048828125, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/norm": 0.00361869831233591, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/mean": 1.5302794054150581e-07, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/std": 7.067774610758166e-06, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/max_abs": 6.008148193359375e-05, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_input_layernorm_weight/norm": 0.01193233402177522, "train/tensor_grad_model_layers_3_input_layernorm_weight/mean": -2.933293581008911e-05, "train/tensor_grad_model_layers_3_input_layernorm_weight/std": 0.0002630412620747739, "train/tensor_grad_model_layers_3_input_layernorm_weight/max_abs": 0.0011444091796875, "train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/norm": 1.3397569099955633, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/mean": -5.597248673439026e-06, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/std": 0.0015105998717445585, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/max_abs": 0.0106201171875, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/norm": 1.3695431714754949, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/mean": 5.347654223442078e-06, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/std": 0.0015439644136133954, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/max_abs": 0.01214599609375, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/norm": 0.02947732664685429, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/mean": -4.059821367263794e-05, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/std": 0.0006524139082055273, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/max_abs": 0.0024566650390625, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/norm": 0.7961176845258154, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/mean": 2.1919608116149902e-05, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/std": 0.0015549180956571612, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/max_abs": 0.01141357421875, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/norm": 0.8217354195429605, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/mean": 4.168599843978883e-06, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/std": 0.001604952290437103, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/max_abs": 0.0146484375, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/norm": 0.003785946264460507, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/mean": -1.1560132406884804e-08, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/std": 7.394430060790757e-06, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/max_abs": 7.200241088867188e-05, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/norm": 0.00380295089392798, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/mean": -1.0211692824668717e-08, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/std": 7.427640735439923e-06, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/max_abs": 6.818771362304688e-05, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_input_layernorm_weight/norm": 0.015525433584795202, "train/tensor_grad_model_layers_2_input_layernorm_weight/mean": -1.8913298845291138e-05, "train/tensor_grad_model_layers_2_input_layernorm_weight/std": 0.00034395847835876135, "train/tensor_grad_model_layers_2_input_layernorm_weight/max_abs": 0.00156402587890625, "train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/norm": 1.6028886468373764, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/mean": -2.7641654014587402e-06, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/std": 0.0018071383851468192, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/max_abs": 0.01165771484375, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/norm": 1.6307447564512847, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/mean": 6.681308150291443e-06, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/std": 0.0018395991081736921, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/max_abs": 0.01409912109375, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/norm": 0.030599392894608033, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/mean": 8.785724639892578e-05, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/std": 0.0006726479946911412, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/max_abs": 0.002288818359375, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/norm": 0.9184200395012729, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/mean": 8.452334441244602e-07, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/std": 0.001793789573474741, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/max_abs": 0.01202392578125, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/norm": 0.9715736829740469, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/mean": 9.313225746154785e-06, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/std": 0.0018976140795652556, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/max_abs": 0.0179443359375, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/norm": 0.004660286237673629, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/mean": 2.700289769563824e-09, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/std": 9.102143103533235e-06, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/max_abs": 0.00010967254638671875, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/norm": 0.005419593448826403, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/mean": -2.140740207323688e-08, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/std": 1.0585167823149843e-05, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/max_abs": 0.00010013580322265625, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_input_layernorm_weight/norm": 0.01827004084958102, "train/tensor_grad_model_layers_1_input_layernorm_weight/mean": 6.294751074165106e-06, "train/tensor_grad_model_layers_1_input_layernorm_weight/std": 0.0004053867194984944, "train/tensor_grad_model_layers_1_input_layernorm_weight/max_abs": 0.0017547607421875, "train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/norm": 2.1033224798165313, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/mean": -2.6656780391931534e-06, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/std": 0.002371735952388378, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/max_abs": 0.0167236328125, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/norm": 2.220554531363239, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/mean": -1.013778273772914e-06, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/std": 0.002505295719311074, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/max_abs": 0.0220947265625, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/norm": 0.047405460099621974, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/mean": -3.2326788641512394e-06, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/std": 0.0010510953850146644, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/max_abs": 0.005035400390625, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/norm": 2.0027208893636077, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/mean": 1.0168878361582756e-07, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/std": 0.003911565519785202, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/max_abs": 0.0361328125, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/norm": 1.942051617141671, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/mean": 6.8140216171741486e-06, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/std": 0.003793071416533045, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/max_abs": 0.0284423828125, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/norm": 0.016645681436829967, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/mean": 3.5588163882493973e-07, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/std": 3.251112734692224e-05, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/max_abs": 0.000301361083984375, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/norm": 0.018046961449750452, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/mean": -3.7724385038018227e-07, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/std": 3.524805191464278e-05, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/max_abs": 0.000621795654296875, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_input_layernorm_weight/norm": 0.03405985944469724, "train/tensor_grad_model_layers_0_input_layernorm_weight/mean": -7.338821887969971e-05, "train/tensor_grad_model_layers_0_input_layernorm_weight/std": 0.000752159456492524, "train/tensor_grad_model_layers_0_input_layernorm_weight/max_abs": 0.003204345703125, "train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_embed_tokens_weight/norm": 3.6260774260900717, "train/tensor_grad_model_embed_tokens_weight/mean": 5.347654223442078e-06, "train/tensor_grad_model_embed_tokens_weight/std": 0.0012535215353413787, "train/tensor_grad_model_embed_tokens_weight/max_abs": 0.1865234375, "train/tensor_grad_model_embed_tokens_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_embed_tokens_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_embed_tokens_weight/norm": 14.4375, "train/tensor_param_model_embed_tokens_weight/mean": 4.220008850097656e-05, "train/tensor_param_model_embed_tokens_weight/std": 0.02001953125, "train/tensor_param_model_embed_tokens_weight/max_abs": 0.0966796875, "train/tensor_param_model_embed_tokens_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_embed_tokens_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/mean": -0.00022792816162109375, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/max_abs": 0.083984375, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/mean": 0.00017547607421875, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/max_abs": 0.08349609375, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/mean": 0.00013065338134765625, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/max_abs": 0.0830078125, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/mean": -3.933906555175781e-05, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/max_abs": 0.08642578125, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_0_mlp_up_proj_weight/mean": -3.5762786865234375e-05, "train/tensor_param_model_layers_0_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_0_mlp_up_proj_weight/max_abs": 0.08984375, "train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_0_mlp_down_proj_weight/mean": 6.818771362304688e-05, "train/tensor_param_model_layers_0_mlp_down_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_0_mlp_down_proj_weight/max_abs": 0.08642578125, "train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_0_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_0_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_0_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/mean": 3.361701965332031e-05, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/max_abs": 0.08447265625, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/mean": -2.8252601623535156e-05, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/max_abs": 0.08837890625, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/mean": -0.0002040863037109375, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/max_abs": 0.08056640625, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/mean": 2.6673078536987305e-06, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/max_abs": 0.076171875, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_1_mlp_up_proj_weight/mean": -6.628036499023438e-05, "train/tensor_param_model_layers_1_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_1_mlp_up_proj_weight/max_abs": 0.080078125, "train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_1_mlp_down_proj_weight/mean": -5.340576171875e-05, "train/tensor_param_model_layers_1_mlp_down_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_1_mlp_down_proj_weight/max_abs": 0.07958984375, "train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_1_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_1_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_1_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/mean": -0.000156402587890625, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/max_abs": 0.0869140625, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/mean": -3.6209821701049805e-06, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/max_abs": 0.08349609375, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/mean": 2.7179718017578125e-05, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/max_abs": 0.08544921875, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/mean": -4.8160552978515625e-05, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/max_abs": 0.08447265625, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_2_mlp_up_proj_weight/mean": 3.7670135498046875e-05, "train/tensor_param_model_layers_2_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_2_mlp_up_proj_weight/max_abs": 0.080078125, "train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_2_mlp_down_proj_weight/mean": -0.00014400482177734375, "train/tensor_param_model_layers_2_mlp_down_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_2_mlp_down_proj_weight/max_abs": 0.09130859375, "train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_2_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_2_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_2_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/mean": -9.632110595703125e-05, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/max_abs": 0.0830078125, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/mean": 7.05718994140625e-05, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/max_abs": 0.0810546875, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/mean": 2.518296241760254e-06, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/max_abs": 0.080078125, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/mean": -0.00020885467529296875, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/max_abs": 0.07958984375, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_3_mlp_up_proj_weight/mean": -0.00010204315185546875, "train/tensor_param_model_layers_3_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_3_mlp_up_proj_weight/max_abs": 0.083984375, "train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_3_mlp_down_proj_weight/mean": -0.00019931793212890625, "train/tensor_param_model_layers_3_mlp_down_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_3_mlp_down_proj_weight/max_abs": 0.08544921875, "train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_3_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_3_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_3_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/mean": -8.153915405273438e-05, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/max_abs": 0.0869140625, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/mean": 5.8650970458984375e-05, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/max_abs": 0.078125, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/norm": 2.546875, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/mean": -8.440017700195312e-05, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/std": 0.0198974609375, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/max_abs": 0.0810546875, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/mean": -9.775161743164062e-05, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/max_abs": 0.091796875, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_4_mlp_up_proj_weight/mean": -1.728534698486328e-05, "train/tensor_param_model_layers_4_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_4_mlp_up_proj_weight/max_abs": 0.08740234375, "train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_4_mlp_down_proj_weight/mean": 4.291534423828125e-05, "train/tensor_param_model_layers_4_mlp_down_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_4_mlp_down_proj_weight/max_abs": 0.0888671875, "train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_4_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_4_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_4_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/mean": -0.00011014938354492188, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/max_abs": 0.08056640625, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/mean": -6.198883056640625e-05, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/max_abs": 0.0859375, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/norm": 2.546875, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/mean": -0.0001392364501953125, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/std": 0.0198974609375, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/max_abs": 0.09033203125, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/mean": 0.00015544891357421875, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/max_abs": 0.0908203125, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_5_mlp_up_proj_weight/mean": -0.0001354217529296875, "train/tensor_param_model_layers_5_mlp_up_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_5_mlp_up_proj_weight/max_abs": 0.09033203125, "train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_mlp_down_proj_weight/norm": 4.40625, "train/tensor_param_model_layers_5_mlp_down_proj_weight/mean": 0.00016880035400390625, "train/tensor_param_model_layers_5_mlp_down_proj_weight/std": 0.0198974609375, "train/tensor_param_model_layers_5_mlp_down_proj_weight/max_abs": 0.087890625, "train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_5_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_5_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_5_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/norm": 2.53125, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/mean": 1.7762184143066406e-05, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/std": 0.019775390625, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/max_abs": 0.08349609375, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/mean": 0.00016307830810546875, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/max_abs": 0.07958984375, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/mean": -0.0002498626708984375, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/max_abs": 0.0791015625, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/norm": 2.546875, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/mean": 0.00021076202392578125, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/std": 0.0198974609375, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/max_abs": 0.0791015625, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_6_mlp_up_proj_weight/mean": 8.916854858398438e-05, "train/tensor_param_model_layers_6_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_6_mlp_up_proj_weight/max_abs": 0.08056640625, "train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_6_mlp_down_proj_weight/mean": 0.000102996826171875, "train/tensor_param_model_layers_6_mlp_down_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_6_mlp_down_proj_weight/max_abs": 0.0830078125, "train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_6_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_6_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_6_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/mean": -0.0002689361572265625, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/max_abs": 0.07861328125, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/mean": 0.00031280517578125, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/max_abs": 0.08056640625, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/mean": -0.00011873245239257812, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/max_abs": 0.08251953125, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/mean": -8.296966552734375e-05, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/max_abs": 0.078125, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_7_mlp_up_proj_weight/mean": -8.249282836914062e-05, "train/tensor_param_model_layers_7_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_7_mlp_up_proj_weight/max_abs": 0.0830078125, "train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_7_mlp_down_proj_weight/mean": 2.753734588623047e-05, "train/tensor_param_model_layers_7_mlp_down_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_7_mlp_down_proj_weight/max_abs": 0.09033203125, "train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_7_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_7_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_7_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/norm": 2.546875, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/mean": 1.0609626770019531e-05, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/std": 0.0198974609375, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/max_abs": 0.08154296875, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/mean": 8.630752563476562e-05, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/max_abs": 0.0810546875, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/norm": 2.53125, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/mean": -0.0002593994140625, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/std": 0.019775390625, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/max_abs": 0.07958984375, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/mean": -1.3589859008789062e-05, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/max_abs": 0.07666015625, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_8_mlp_up_proj_weight/mean": -0.00018310546875, "train/tensor_param_model_layers_8_mlp_up_proj_weight/std": 0.0198974609375, "train/tensor_param_model_layers_8_mlp_up_proj_weight/max_abs": 0.1025390625, "train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_mlp_down_proj_weight/norm": 4.40625, "train/tensor_param_model_layers_8_mlp_down_proj_weight/mean": -0.0002040863037109375, "train/tensor_param_model_layers_8_mlp_down_proj_weight/std": 0.0198974609375, "train/tensor_param_model_layers_8_mlp_down_proj_weight/max_abs": 0.0830078125, "train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_8_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_8_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_8_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_norm_weight/norm": 11.3125, "train/tensor_param_model_norm_weight/mean": 1.0, "train/tensor_param_model_norm_weight/std": 0.0, "train/tensor_param_model_norm_weight/max_abs": 1.0, "train/tensor_param_model_norm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_norm_weight/frac_near_user_limit": 0.0} +{"step": 40, "epoch": 0.053926525109538256, "timestamp": 1786254905.1471906, "loss": 102.05034790039062, "grad_norm": 24.625, "learning_rate": 0.0005, "train/total_time_seconds": 41.32444467023015, "train/time_per_step_avg": 1.0331111167557538, "train/epoch_time_elapsed": 80.11485634371638, "train/estimated_remaining_minutes": 12.225148214943088} +{"step": 50, "epoch": 0.06740815638692282, "timestamp": 1786254932.9985719, "eval_loss": 5.665860652923584, "eval_runtime": 9.5524, "eval_samples_per_second": 997.346, "eval_steps_per_second": 12.562, "train/total_time_seconds": 50.11429492011666, "train/time_per_step_avg": 1.0022858984023333, "train/epoch_time_elapsed": 107.96623866260052, "train/estimated_remaining_minutes": 11.693335481360554} +{"step": 60, "epoch": 0.08088978766430738, "timestamp": 1786254951.3890784, "loss": 90.97202758789062, "grad_norm": 18.125, "learning_rate": 0.0005, "train/total_time_seconds": 58.94606887176633, "train/time_per_step_avg": 0.9824344811961054, "train/epoch_time_elapsed": 126.35674532875419, "train/estimated_remaining_minutes": 11.297996533755214} +{"step": 80, "epoch": 0.10785305021907651, "timestamp": 1786254989.0803852, "loss": 83.10302734375, "grad_norm": 40.75, "learning_rate": 0.0005, "train/total_time_seconds": 77.25812346488237, "train/time_per_step_avg": 0.9657265433110297, "train/epoch_time_elapsed": 164.04805217683315, "train/estimated_remaining_minutes": 10.783946400306498} +{"step": 100, "epoch": 0.13481631277384565, "timestamp": 1786255023.7892716, "loss": 77.51190185546875, "grad_norm": 29.625, "learning_rate": 0.0005, "train/total_time_seconds": 92.4002431370318, "train/time_per_step_avg": 0.9240024313703179, "train/epoch_time_elapsed": 198.7569378837943, "train/estimated_remaining_minutes": 10.01002633984511} +{"step": 100, "epoch": 0.13481631277384565, "timestamp": 1786255032.9020689, "eval_loss": 4.689935684204102, "eval_runtime": 9.1111, "eval_samples_per_second": 1045.652, "eval_steps_per_second": 13.171, "train/total_time_seconds": 92.4002431370318, "train/time_per_step_avg": 0.9240024313703179, "train/epoch_time_elapsed": 207.8697351180017, "train/estimated_remaining_minutes": 10.01002633984511} +{"step": 120, "epoch": 0.16177957532861476, "timestamp": 1786255064.323373, "loss": 73.29612426757812, "grad_norm": 19.5, "learning_rate": 0.0005, "train/total_time_seconds": 104.50122643262148, "train/time_per_step_avg": 0.812197428420186, "train/epoch_time_elapsed": 239.29103926569223, "train/estimated_remaining_minutes": 9.14385731285438} +{"step": 140, "epoch": 0.1887428378833839, "timestamp": 1786255101.9086154, "loss": 70.37640380859375, "grad_norm": 25.25, "learning_rate": 0.0005, "train/total_time_seconds": 122.9972688369453, "train/time_per_step_avg": 0.8167282416671514, "train/epoch_time_elapsed": 276.87628524005413, "train/estimated_remaining_minutes": 8.931944522682931} +{"step": 150, "epoch": 0.20222446916076844, "timestamp": 1786255130.0236912, "eval_loss": 4.263797760009766, "eval_runtime": 9.4931, "eval_samples_per_second": 1003.568, "eval_steps_per_second": 12.641, "train/total_time_seconds": 131.93278025463223, "train/time_per_step_avg": 0.8181848533451557, "train/epoch_time_elapsed": 304.9913580864668, "train/estimated_remaining_minutes": 8.795518683642149} +{"step": 160, "epoch": 0.21570610043815303, "timestamp": 1786255148.6720192, "loss": 68.2427978515625, "grad_norm": 14.625, "learning_rate": 0.0005, "train/total_time_seconds": 141.14684723690152, "train/time_per_step_avg": 0.822007783651352, "train/epoch_time_elapsed": 323.6396882608533, "train/estimated_remaining_minutes": 8.674649986434572} +{"step": 180, "epoch": 0.24266936299292213, "timestamp": 1786255185.931212, "loss": 66.13766479492188, "grad_norm": 13.0625, "learning_rate": 0.0005, "train/total_time_seconds": 159.10167576372623, "train/time_per_step_avg": 0.8184355229884386, "train/epoch_time_elapsed": 360.8988819755614, "train/estimated_remaining_minutes": 8.397032887529996} +{"step": 200, "epoch": 0.2696326255476913, "timestamp": 1786255222.6511283, "loss": 63.81178588867188, "grad_norm": 18.5, "learning_rate": 0.0005, "train/total_time_seconds": 176.25136033818126, "train/time_per_step_avg": 0.8385111720114946, "train/epoch_time_elapsed": 397.61879746988416, "train/estimated_remaining_minutes": 8.078187348833307} +{"step": 200, "epoch": 0.2696326255476913, "timestamp": 1786255232.0393953, "eval_loss": 3.9343974590301514, "eval_runtime": 9.3868, "eval_samples_per_second": 1014.939, "eval_steps_per_second": 12.784, "train/total_time_seconds": 176.25136033818126, "train/time_per_step_avg": 0.8385111720114946, "train/epoch_time_elapsed": 407.00706396251917, "train/estimated_remaining_minutes": 8.078187348833307} +{"step": 220, "epoch": 0.2965958881024604, "timestamp": 1786255269.9923177, "loss": 62.16768188476563, "grad_norm": 16.125, "learning_rate": 0.0005, "train/total_time_seconds": 194.7538175098598, "train/time_per_step_avg": 0.9025259107723832, "train/epoch_time_elapsed": 444.9599856287241, "train/estimated_remaining_minutes": 7.819660854562553} +{"step": 240, "epoch": 0.3235591506572295, "timestamp": 1786255307.355241, "loss": 61.1105712890625, "grad_norm": 21.25, "learning_rate": 0.0005, "train/total_time_seconds": 212.70383574068546, "train/time_per_step_avg": 0.8970656690374017, "train/epoch_time_elapsed": 482.3229112736881, "train/estimated_remaining_minutes": 7.5332608491492765} +{"step": 250, "epoch": 0.33704078193461406, "timestamp": 1786255336.121311, "eval_loss": 3.748317241668701, "eval_runtime": 9.4777, "eval_samples_per_second": 1005.204, "eval_steps_per_second": 12.661, "train/total_time_seconds": 222.2051606848836, "train/time_per_step_avg": 0.9027238043025136, "train/epoch_time_elapsed": 511.08898059278727, "train/estimated_remaining_minutes": 7.40683868949612} +{"step": 260, "epoch": 0.3505224132119987, "timestamp": 1786255353.3154516, "loss": 60.05653686523438, "grad_norm": 42.25, "learning_rate": 0.0005, "train/total_time_seconds": 229.90829008817673, "train/time_per_step_avg": 0.8876144285127521, "train/epoch_time_elapsed": 528.2831210829318, "train/estimated_remaining_minutes": 7.221478342513243} +{"step": 280, "epoch": 0.3774856757667678, "timestamp": 1786255390.8281882, "loss": 59.25806884765625, "grad_norm": 27.5, "learning_rate": 0.0005, "train/total_time_seconds": 248.29544523730874, "train/time_per_step_avg": 0.8919376947358251, "train/epoch_time_elapsed": 565.7958568930626, "train/estimated_remaining_minutes": 6.9463606703294705} +{"step": 300, "epoch": 0.4044489383215369, "timestamp": 1786255428.5006192, "loss": 58.334844970703124, "grad_norm": 14.375, "learning_rate": 0.0005, "train/total_time_seconds": 266.7115136086941, "train/time_per_step_avg": 0.9046015327051282, "train/epoch_time_elapsed": 603.468289192766, "train/estimated_remaining_minutes": 6.667787840217352} +{"step": 300, "epoch": 0.4044489383215369, "timestamp": 1786255438.167693, "eval_loss": 3.6137855052948, "eval_runtime": 9.6654, "eval_samples_per_second": 985.685, "eval_steps_per_second": 12.415, "train/total_time_seconds": 266.7115136086941, "train/time_per_step_avg": 0.9046015327051282, "train/epoch_time_elapsed": 613.1353628858924, "train/estimated_remaining_minutes": 6.667787840217352} +{"step": 320, "epoch": 0.43141220087630605, "timestamp": 1786255475.8042827, "loss": 57.341180419921876, "grad_norm": 32.75, "learning_rate": 0.0005, "train/total_time_seconds": 284.8533755801618, "train/time_per_step_avg": 0.9009955807030201, "train/epoch_time_elapsed": 650.7719515115023, "train/estimated_remaining_minutes": 6.379528723930707} +{"step": 340, "epoch": 0.45837546343107516, "timestamp": 1786255511.6513586, "loss": 56.699591064453124, "grad_norm": 12.5625, "learning_rate": 0.0005, "train/total_time_seconds": 301.7134180255234, "train/time_per_step_avg": 0.8900958228483796, "train/epoch_time_elapsed": 686.6190287470818, "train/estimated_remaining_minutes": 6.0638481073757164} +{"step": 350, "epoch": 0.4718570947084597, "timestamp": 1786255540.1244693, "eval_loss": 3.492727518081665, "eval_runtime": 9.4437, "eval_samples_per_second": 1008.824, "eval_steps_per_second": 12.707, "train/total_time_seconds": 311.0745892226696, "train/time_per_step_avg": 0.8886942853778601, "train/epoch_time_elapsed": 715.0921384617686, "train/estimated_remaining_minutes": 5.925230270907992} +{"step": 360, "epoch": 0.48533872598584427, "timestamp": 1786255558.9333723, "loss": 55.840753173828126, "grad_norm": 21.5, "learning_rate": 0.0005, "train/total_time_seconds": 320.1871528066695, "train/time_per_step_avg": 0.9027886271849275, "train/epoch_time_elapsed": 733.9010420665145, "train/estimated_remaining_minutes": 5.781156925675977} +{"step": 380, "epoch": 0.5123019885406134, "timestamp": 1786255596.6807182, "loss": 55.377203369140624, "grad_norm": 14.5625, "learning_rate": 0.0005, "train/total_time_seconds": 338.4666022621095, "train/time_per_step_avg": 0.9017115702480077, "train/epoch_time_elapsed": 771.6483867913485, "train/estimated_remaining_minutes": 5.492659773551777} +{"step": 400, "epoch": 0.5392652510953826, "timestamp": 1786255633.9909048, "loss": 54.936090087890626, "grad_norm": 14.6875, "learning_rate": 0.0005, "train/total_time_seconds": 356.61116948351264, "train/time_per_step_avg": 0.8989965587481856, "train/epoch_time_elapsed": 808.9585740156472, "train/estimated_remaining_minutes": 5.200579554967892} +{"step": 400, "epoch": 0.5392652510953826, "timestamp": 1786255643.176145, "eval_loss": 3.41583251953125, "eval_runtime": 9.1837, "eval_samples_per_second": 1037.382, "eval_steps_per_second": 13.067, "train/total_time_seconds": 356.61116948351264, "train/time_per_step_avg": 0.8989965587481856, "train/epoch_time_elapsed": 818.1438147202134, "train/estimated_remaining_minutes": 5.200579554967892} +{"step": 420, "epoch": 0.5662285136501517, "timestamp": 1786255680.8474326, "loss": 54.31259765625, "grad_norm": 22.0, "learning_rate": 0.0005, "train/total_time_seconds": 374.4849713034928, "train/time_per_step_avg": 0.8963159572333097, "train/epoch_time_elapsed": 855.8150995858014, "train/estimated_remaining_minutes": 4.903969862307644} +{"step": 440, "epoch": 0.5931917762049208, "timestamp": 1786255718.4085093, "loss": 53.894476318359374, "grad_norm": 22.5, "learning_rate": 0.0005, "train/total_time_seconds": 392.4683515615761, "train/time_per_step_avg": 0.9075493353605271, "train/epoch_time_elapsed": 893.3761757835746, "train/estimated_remaining_minutes": 4.608529885760932} +{"step": 450, "epoch": 0.6066734074823054, "timestamp": 1786255746.9943676, "eval_loss": 3.3446545600891113, "eval_runtime": 9.6871, "eval_samples_per_second": 983.477, "eval_steps_per_second": 12.388, "train/total_time_seconds": 401.84830053150654, "train/time_per_step_avg": 0.9077371130883694, "train/epoch_time_elapsed": 921.962036266923, "train/estimated_remaining_minutes": 4.464981117016739} +{"step": 460, "epoch": 0.6201550387596899, "timestamp": 1786255765.591648, "loss": 53.53094482421875, "grad_norm": 21.875, "learning_rate": 0.0005, "train/total_time_seconds": 410.88976757600904, "train/time_per_step_avg": 0.9070261476933956, "train/epoch_time_elapsed": 940.5593165978789, "train/estimated_remaining_minutes": 4.317320021631979} +{"step": 480, "epoch": 0.647118301314459, "timestamp": 1786255801.5716844, "loss": 53.14410400390625, "grad_norm": 20.625, "learning_rate": 0.0005, "train/total_time_seconds": 427.66226352751255, "train/time_per_step_avg": 0.8919566126540304, "train/epoch_time_elapsed": 976.5393532849848, "train/estimated_remaining_minutes": 4.00933372057043} +{"step": 500, "epoch": 0.6740815638692281, "timestamp": 1786255838.9319963, "loss": 52.86866455078125, "grad_norm": 19.75, "learning_rate": 0.0005, "train/total_time_seconds": 445.7218977920711, "train/time_per_step_avg": 0.8911072830855846, "train/epoch_time_elapsed": 1013.899667005986, "train/estimated_remaining_minutes": 3.7143491482672593} +{"step": 500, "epoch": 0.6740815638692281, "timestamp": 1786255848.527109, "eval_loss": 3.2967636585235596, "eval_runtime": 9.5934, "eval_samples_per_second": 993.08, "eval_steps_per_second": 12.509, "train/total_time_seconds": 445.7218977920711, "train/time_per_step_avg": 0.8911072830855846, "train/epoch_time_elapsed": 1023.4947780184448, "train/estimated_remaining_minutes": 3.7143491482672593} +{"step": 520, "epoch": 0.7010448264239973, "timestamp": 1786255886.4791985, "loss": 52.4890625, "grad_norm": 31.0, "learning_rate": 0.0005, "train/total_time_seconds": 464.1531522870064, "train/time_per_step_avg": 0.8966818098351359, "train/epoch_time_elapsed": 1061.446867160499, "train/estimated_remaining_minutes": 3.421641827756778} +{"step": 540, "epoch": 0.7280080889787665, "timestamp": 1786255924.1253083, "loss": 52.321435546875, "grad_norm": 16.875, "learning_rate": 0.0005, "train/total_time_seconds": 482.53898264840245, "train/time_per_step_avg": 0.9007063108682632, "train/epoch_time_elapsed": 1099.0929772891104, "train/estimated_remaining_minutes": 3.1275674801285342} +{"step": 550, "epoch": 0.741489720256151, "timestamp": 1786255951.6409235, "eval_loss": 3.250918388366699, "eval_runtime": 9.1469, "eval_samples_per_second": 1041.558, "eval_steps_per_second": 13.119, "train/total_time_seconds": 490.9854051284492, "train/time_per_step_avg": 0.8913710459694266, "train/epoch_time_elapsed": 1126.6085901521146, "train/estimated_remaining_minutes": 2.975669121990601} +{"step": 560, "epoch": 0.7549713515335356, "timestamp": 1786255971.0950038, "loss": 52.020050048828125, "grad_norm": 18.5, "learning_rate": 0.0005, "train/total_time_seconds": 500.46789940074086, "train/time_per_step_avg": 0.8957813182473182, "train/epoch_time_elapsed": 1146.0626736655831, "train/estimated_remaining_minutes": 2.8300268120875227} +{"step": 580, "epoch": 0.7819346140883047, "timestamp": 1786256008.5453694, "loss": 51.72322387695313, "grad_norm": 23.875, "learning_rate": 0.0005, "train/total_time_seconds": 518.4827314689755, "train/time_per_step_avg": 0.9082046794146299, "train/epoch_time_elapsed": 1183.5130392313004, "train/estimated_remaining_minutes": 2.532817941084076} +{"step": 600, "epoch": 0.8088978766430738, "timestamp": 1786256046.9037464, "loss": 51.40700073242188, "grad_norm": 17.125, "learning_rate": 0.0005, "train/total_time_seconds": 537.0862419344485, "train/time_per_step_avg": 0.9136434414237737, "train/epoch_time_elapsed": 1221.8714155852795, "train/estimated_remaining_minutes": 2.2378593413935355} +{"step": 600, "epoch": 0.8088978766430738, "timestamp": 1786256056.4047332, "eval_loss": 3.21195650100708, "eval_runtime": 9.4995, "eval_samples_per_second": 1002.9, "eval_steps_per_second": 12.632, "train/total_time_seconds": 537.0862419344485, "train/time_per_step_avg": 0.9136434414237737, "train/epoch_time_elapsed": 1231.3724030666053, "train/estimated_remaining_minutes": 2.2378593413935355} +{"step": 620, "epoch": 0.8358611391978429, "timestamp": 1786256093.7350392, "loss": 51.1299072265625, "grad_norm": 18.125, "learning_rate": 0.0005, "train/total_time_seconds": 554.5250961445272, "train/time_per_step_avg": 0.9037194385752082, "train/epoch_time_elapsed": 1268.702707707882, "train/estimated_remaining_minutes": 1.9378565187846382} +{"step": 640, "epoch": 0.8628244017526121, "timestamp": 1786256130.7095737, "loss": 50.86155700683594, "grad_norm": 18.875, "learning_rate": 0.0005, "train/total_time_seconds": 571.9027761295438, "train/time_per_step_avg": 0.8936379348114133, "train/epoch_time_elapsed": 1305.6772438436747, "train/estimated_remaining_minutes": 1.6382631607877556} +{"step": 650, "epoch": 0.8763060330299967, "timestamp": 1786256159.5274856, "eval_loss": 3.1778173446655273, "eval_runtime": 9.5123, "eval_samples_per_second": 1001.54, "eval_steps_per_second": 12.615, "train/total_time_seconds": 581.2734086625278, "train/time_per_step_avg": 0.902880035340786, "train/epoch_time_elapsed": 1334.4951537139714, "train/estimated_remaining_minutes": 1.4904446375962253} +{"step": 660, "epoch": 0.8897876643073812, "timestamp": 1786256178.165015, "loss": 50.78846435546875, "grad_norm": 30.125, "learning_rate": 0.0005, "train/total_time_seconds": 590.354661539197, "train/time_per_step_avg": 0.8988676213845611, "train/epoch_time_elapsed": 1353.1326842531562, "train/estimated_remaining_minutes": 1.3417151398618115} +{"step": 680, "epoch": 0.9167509268621503, "timestamp": 1786256216.171869, "loss": 50.50668640136719, "grad_norm": 23.0, "learning_rate": 0.0005, "train/total_time_seconds": 608.6652480624616, "train/time_per_step_avg": 0.9018251659348607, "train/epoch_time_elapsed": 1391.1395382620394, "train/estimated_remaining_minutes": 1.0442786118718703} +{"step": 700, "epoch": 0.9437141894169194, "timestamp": 1786256251.677505, "loss": 50.196780395507815, "grad_norm": 18.25, "learning_rate": 0.0005, "train/total_time_seconds": 624.9607063494623, "train/time_per_step_avg": 0.8787446441501379, "train/epoch_time_elapsed": 1426.645174805075, "train/estimated_remaining_minutes": 0.7440008408922171} +{"step": 700, "epoch": 0.9437141894169194, "timestamp": 1786256261.0661898, "eval_loss": 3.135035991668701, "eval_runtime": 9.3872, "eval_samples_per_second": 1014.894, "eval_steps_per_second": 12.783, "train/total_time_seconds": 624.9607063494623, "train/time_per_step_avg": 0.8787446441501379, "train/epoch_time_elapsed": 1436.0338580720127, "train/estimated_remaining_minutes": 0.7440008408922171} +{"step": 720, "epoch": 0.9706774519716885, "timestamp": 1786256298.7307463, "loss": 49.94974365234375, "grad_norm": 21.75, "learning_rate": 0.0005, "train/total_time_seconds": 643.4756709970534, "train/time_per_step_avg": 0.8895057485252619, "train/epoch_time_elapsed": 1473.6984132528305, "train/estimated_remaining_minutes": 0.44685810485906485} +{"step": 740, "epoch": 0.9976407145264578, "timestamp": 1786256336.3365912, "loss": 49.74850769042969, "grad_norm": 25.125, "learning_rate": 0.0005, "train/total_time_seconds": 661.7013132050633, "train/time_per_step_avg": 0.8979853707551956, "train/epoch_time_elapsed": 1511.3042608723044, "train/estimated_remaining_minutes": 0.1490318272984377} +{"step": 750, "epoch": 1.0107853050219076, "timestamp": 1786256369.8029494, "eval_loss": 3.108103036880493, "eval_runtime": 9.4956, "eval_samples_per_second": 1003.31, "eval_steps_per_second": 12.637, "train/total_time_seconds": 676.0385475568473, "train/time_per_step_avg": 0.9476513889431953, "train/epoch_time_elapsed": 24.854905635118484, "train/estimated_remaining_minutes": 0.0, "train/global/act/norm": 150185.3480440026, "train/global/act/mean": -0.42774814351682083, "train/global/act/std": 1.2584756013799252, "train/global/act/max_abs": 11.8125, "train/global/act/frac_near_dtype_limit": 0.0, "train/global/act/frac_near_user_limit": 0.0, "train/global/grad/norm": 4.998155692517693, "train/global/grad/mean": 1.7156429580753228e-07, "train/global/grad/std": 0.0008832509986677723, "train/global/grad/max_abs": 0.05859375, "train/global/grad/frac_near_dtype_limit": 0.0, "train/global/grad/frac_near_user_limit": 0.0, "train/global/param/norm": 75.44594149383137, "train/global/param/mean": 0.0009513001034806474, "train/global/param/std": 0.05331836449904397, "train/global/param/max_abs": 1.0, "train/global/param/frac_near_dtype_limit": 0.0, "train/global/param/frac_near_user_limit": 0.0, "train/layer__model_layers_4/param/norm": 18.830361641010775, "train/layer__model_layers_4/param/mean": 0.0016009677404918462, "train/layer__model_layers_4/param/std": 0.04647544995312572, "train/layer__model_layers_4/param/max_abs": 1.0, "train/layer__model_layers_4/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_4/param/frac_near_user_limit": 0.0, "train/layer__model_layers_7/param/norm": 19.057209819219338, "train/layer__model_layers_7/param/mean": 0.001528687856498635, "train/layer__model_layers_7/param/std": 0.04701331013729768, "train/layer__model_layers_7/param/max_abs": 1.0, "train/layer__model_layers_7/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_7/param/frac_near_user_limit": 0.0, "train/layer__model_layers_6/param/norm": 19.296900303627133, "train/layer__model_layers_6/param/mean": 0.0016313573685525545, "train/layer__model_layers_6/param/std": 0.047603568559157615, "train/layer__model_layers_6/param/max_abs": 1.0, "train/layer__model_layers_6/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_6/param/frac_near_user_limit": 0.0, "train/layer_model_layers_6/act/norm": 20266.952184443417, "train/layer_model_layers_6/act/mean": -0.005417345808102534, "train/layer_model_layers_6/act/std": 0.6138518434558304, "train/layer_model_layers_6/act/max_abs": 6.03125, "train/layer_model_layers_6/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_6/act/frac_near_user_limit": 0.0, "train/layer_model_layers_6/grad/norm": 1.1219461188048783, "train/layer_model_layers_6/grad/mean": 1.0562621859046114e-06, "train/layer_model_layers_6/grad/std": 0.0006925265544513297, "train/layer_model_layers_6/grad/max_abs": 0.0098876953125, "train/layer_model_layers_6/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_6/grad/frac_near_user_limit": 0.0, "train/layer_model_layers_7/act/norm": 20479.295498580417, "train/layer_model_layers_7/act/mean": -0.004186410170335036, "train/layer_model_layers_7/act/std": 0.6202687217424423, "train/layer_model_layers_7/act/max_abs": 7.0, "train/layer_model_layers_7/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_7/act/frac_near_user_limit": 0.0, "train/layer_model_layers_7/grad/norm": 0.9808154580741602, "train/layer_model_layers_7/grad/mean": -6.818339608568995e-08, "train/layer_model_layers_7/grad/std": 0.0006054649694080247, "train/layer_model_layers_7/grad/max_abs": 0.0093994140625, "train/layer_model_layers_7/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_7/grad/frac_near_user_limit": 0.0, "train/layer__model_layers_8/param/norm": 19.81155114052784, "train/layer__model_layers_8/param/mean": 0.0013863359710169657, "train/layer__model_layers_8/param/std": 0.04890227832752276, "train/layer__model_layers_8/param/max_abs": 1.0, "train/layer__model_layers_8/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_8/param/frac_near_user_limit": 0.0, "train/layer_model_layers_4/act/norm": 18073.148360412022, "train/layer_model_layers_4/act/mean": 0.011836038185999943, "train/layer_model_layers_4/act/std": 0.5473840823442112, "train/layer_model_layers_4/act/max_abs": 5.21875, "train/layer_model_layers_4/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_4/act/frac_near_user_limit": 0.0, "train/layer_model_layers_4/grad/norm": 1.0366429293152521, "train/layer_model_layers_4/grad/mean": 1.076114903002187e-06, "train/layer_model_layers_4/grad/std": 0.0006396265473420889, "train/layer_model_layers_4/grad/max_abs": 0.00689697265625, "train/layer_model_layers_4/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_4/grad/frac_near_user_limit": 0.0, "train/layer_model_layers_8/act/norm": 22737.438562436826, "train/layer_model_layers_8/act/mean": -0.0016579261192908655, "train/layer_model_layers_8/act/std": 0.6887773225057738, "train/layer_model_layers_8/act/max_abs": 6.375, "train/layer_model_layers_8/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_8/act/frac_near_user_limit": 0.0, "train/layer_model_layers_8/grad/norm": 0.936803955234771, "train/layer_model_layers_8/grad/mean": 1.9631800408854315e-07, "train/layer_model_layers_8/grad/std": 0.0005782672798617279, "train/layer_model_layers_8/grad/max_abs": 0.009765625, "train/layer_model_layers_8/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_8/grad/frac_near_user_limit": 0.0, "train/layer_model_layers_5/act/norm": 18287.03101790653, "train/layer_model_layers_5/act/mean": 0.005246024865370531, "train/layer_model_layers_5/act/std": 0.5537264622185337, "train/layer_model_layers_5/act/max_abs": 6.0, "train/layer_model_layers_5/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_5/act/frac_near_user_limit": 0.0, "train/layer_model_layers_5/grad/norm": 1.1070405366170881, "train/layer_model_layers_5/grad/mean": 3.8479168589885073e-07, "train/layer_model_layers_5/grad/std": 0.0006830734147017943, "train/layer_model_layers_5/grad/max_abs": 0.01092529296875, "train/layer_model_layers_5/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_5/grad/frac_near_user_limit": 0.0, "train/layer__model_layers_1/param/norm": 18.983731984995075, "train/layer__model_layers_1/param/mean": 0.0015187018003181064, "train/layer__model_layers_1/param/std": 0.046850783894402705, "train/layer__model_layers_1/param/max_abs": 1.0, "train/layer__model_layers_1/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_1/param/frac_near_user_limit": 0.0, "train/layer_model_layers_3/act/norm": 19206.574143126698, "train/layer_model_layers_3/act/mean": 0.0260999844624446, "train/layer_model_layers_3/act/std": 0.5813081153399244, "train/layer_model_layers_3/act/max_abs": 6.21875, "train/layer_model_layers_3/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_3/act/frac_near_user_limit": 0.0, "train/layer_model_layers_3/grad/norm": 1.1688936654531, "train/layer_model_layers_3/grad/mean": 1.3687615914854357e-06, "train/layer_model_layers_3/grad/std": 0.0007216427637524477, "train/layer_model_layers_3/grad/max_abs": 0.00823974609375, "train/layer_model_layers_3/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_3/grad/frac_near_user_limit": 0.0, "train/layer_model_layers_1/act/norm": 17389.814008941397, "train/layer_model_layers_1/act/mean": 0.005572208991417518, "train/layer_model_layers_1/act/std": 0.5266370859241487, "train/layer_model_layers_1/act/max_abs": 5.5, "train/layer_model_layers_1/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_1/act/frac_near_user_limit": 0.0, "train/layer_model_layers_1/grad/norm": 1.748869284242822, "train/layer_model_layers_1/grad/mean": 1.9777443967527144e-06, "train/layer_model_layers_1/grad/std": 0.0010789055805118235, "train/layer_model_layers_1/grad/max_abs": 0.01153564453125, "train/layer_model_layers_1/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_1/grad/frac_near_user_limit": 0.0, "train/layer__model_layers_3/param/norm": 19.16069101254245, "train/layer__model_layers_3/param/mean": 0.0015184280466921801, "train/layer__model_layers_3/param/std": 0.04727126222105599, "train/layer__model_layers_3/param/max_abs": 1.0, "train/layer__model_layers_3/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_3/param/frac_near_user_limit": 0.0, "train/layer__model_layers_2/param/norm": 19.07513669266226, "train/layer__model_layers_2/param/mean": 0.0015293722405634506, "train/layer__model_layers_2/param/std": 0.04707263888163651, "train/layer__model_layers_2/param/max_abs": 1.0, "train/layer__model_layers_2/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_2/param/frac_near_user_limit": 0.0, "train/layer__model_layers_5/param/norm": 18.903182860714885, "train/layer__model_layers_5/param/mean": 0.001528687856498635, "train/layer__model_layers_5/param/std": 0.046641797054845655, "train/layer__model_layers_5/param/max_abs": 1.0, "train/layer__model_layers_5/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_5/param/frac_near_user_limit": 0.0, "train/layer_model_layers_0/act/norm": 17591.951382313935, "train/layer_model_layers_0/act/mean": -0.0008019591824939618, "train/layer_model_layers_0/act/std": 0.5327684547972705, "train/layer_model_layers_0/act/max_abs": 6.21875, "train/layer_model_layers_0/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_0/act/frac_near_user_limit": 0.0, "train/layer_model_layers_0/grad/norm": 3.358869449977567, "train/layer_model_layers_0/grad/mean": 8.951448334174666e-07, "train/layer_model_layers_0/grad/std": 0.002073075775157821, "train/layer_model_layers_0/grad/max_abs": 0.0286865234375, "train/layer_model_layers_0/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_0/grad/frac_near_user_limit": 0.0, "train/layer__model_layers_0/param/norm": 18.78089251947641, "train/layer__model_layers_0/param/mean": 0.0015775290740633531, "train/layer__model_layers_0/param/std": 0.046327244641872406, "train/layer__model_layers_0/param/max_abs": 1.0, "train/layer__model_layers_0/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_0/param/frac_near_user_limit": 0.0, "train/layer_model_layers_2/act/norm": 18632.33940555953, "train/layer_model_layers_2/act/mean": 0.008575276722415136, "train/layer_model_layers_2/act/std": 0.5642383070261626, "train/layer_model_layers_2/act/max_abs": 5.25, "train/layer_model_layers_2/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_2/act/frac_near_user_limit": 0.0, "train/layer_model_layers_2/grad/norm": 1.3519265251055885, "train/layer_model_layers_2/grad/mean": 2.437561762035358e-06, "train/layer_model_layers_2/grad/std": 0.0008341103202784663, "train/layer_model_layers_2/grad/max_abs": 0.007598876953125, "train/layer_model_layers_2/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_2/grad/frac_near_user_limit": 0.0, "train/tensor_act_model_embed_tokens/norm": 686.0245066748595, "train/tensor_act_model_embed_tokens/mean": 0.0006635189056396484, "train/tensor_act_model_embed_tokens/std": 0.07476825003052505, "train/tensor_act_model_embed_tokens/max_abs": 0.25, "train/tensor_act_model_embed_tokens/frac_near_dtype_limit": 0.0, "train/tensor_act_model_embed_tokens/frac_near_user_limit": 0.0, "train/tensor_act_model_rotary_emb/norm": 3632.2067871093745, "train/tensor_act_model_rotary_emb/mean": 0.333984375, "train/tensor_act_model_rotary_emb/std": 0.71875, "train/tensor_act_model_rotary_emb/max_abs": 1.0, "train/tensor_act_model_rotary_emb/frac_near_dtype_limit": 0.0, "train/tensor_act_model_rotary_emb/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_input_layernorm/norm": 9158.321228096187, "train/tensor_act_model_layers_0_input_layernorm/mean": 0.009479522705078125, "train/tensor_act_model_layers_0_input_layernorm/std": 1.000000133579297, "train/tensor_act_model_layers_0_input_layernorm/max_abs": 3.71875, "train/tensor_act_model_layers_0_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_q_proj/norm": 8242.797600267277, "train/tensor_act_model_layers_0_self_attn_q_proj/mean": -0.017574310302734375, "train/tensor_act_model_layers_0_self_attn_q_proj/std": 0.8996636131643495, "train/tensor_act_model_layers_0_self_attn_q_proj/max_abs": 6.21875, "train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_k_proj/norm": 7571.323294073769, "train/tensor_act_model_layers_0_self_attn_k_proj/mean": -0.012340545654296875, "train/tensor_act_model_layers_0_self_attn_k_proj/std": 0.8269067291263272, "train/tensor_act_model_layers_0_self_attn_k_proj/max_abs": 4.3125, "train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_v_proj/norm": 1815.8166580376064, "train/tensor_act_model_layers_0_self_attn_v_proj/mean": -0.0010931491851806638, "train/tensor_act_model_layers_0_self_attn_v_proj/std": 0.198242266075022, "train/tensor_act_model_layers_0_self_attn_v_proj/max_abs": 0.98828125, "train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_o_proj/norm": 139.19229095220038, "train/tensor_act_model_layers_0_self_attn_o_proj/mean": -0.0005276203155517578, "train/tensor_act_model_layers_0_self_attn_o_proj/std": 0.01518657306789068, "train/tensor_act_model_layers_0_self_attn_o_proj/max_abs": 0.224609375, "train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn/norm": 139.19229095220038, "train/tensor_act_model_layers_0_self_attn/mean": -0.0005276203155517578, "train/tensor_act_model_layers_0_self_attn/std": 0.01518657306789068, "train/tensor_act_model_layers_0_self_attn/max_abs": 0.224609375, "train/tensor_act_model_layers_0_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_post_attention_layernorm/norm": 9158.038940433684, "train/tensor_act_model_layers_0_post_attention_layernorm/mean": 0.0021638870239257812, "train/tensor_act_model_layers_0_post_attention_layernorm/std": 1.0000004657698636, "train/tensor_act_model_layers_0_post_attention_layernorm/max_abs": 4.53125, "train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_mlp_up_proj/norm": 3148.9947247149603, "train/tensor_act_model_layers_0_mlp_up_proj/mean": 0.003276824951171875, "train/tensor_act_model_layers_0_mlp_up_proj/std": 0.19842569797190573, "train/tensor_act_model_layers_0_mlp_up_proj/max_abs": 1.1328125, "train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_mlp_down_proj/norm": 1014.9781179029341, "train/tensor_act_model_layers_0_mlp_down_proj/mean": 9.156763553619385e-06, "train/tensor_act_model_layers_0_mlp_down_proj/std": 0.11087058315602684, "train/tensor_act_model_layers_0_mlp_down_proj/max_abs": 0.60546875, "train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_mlp/norm": 1014.9781179029341, "train/tensor_act_model_layers_0_mlp/mean": 9.156763553619385e-06, "train/tensor_act_model_layers_0_mlp/std": 0.11087058315602684, "train/tensor_act_model_layers_0_mlp/max_abs": 0.60546875, "train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0/norm": 1072.8987485266305, "train/tensor_act_model_layers_0/mean": 0.0001455782912671566, "train/tensor_act_model_layers_0/std": 0.11718775424792081, "train/tensor_act_model_layers_0/max_abs": 0.72265625, "train/tensor_act_model_layers_0/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_input_layernorm/norm": 9158.515502939657, "train/tensor_act_model_layers_1_input_layernorm/mean": 0.003910541534423829, "train/tensor_act_model_layers_1_input_layernorm/std": 1.0000012806912975, "train/tensor_act_model_layers_1_input_layernorm/max_abs": 4.875, "train/tensor_act_model_layers_1_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_q_proj/norm": 7272.671681066942, "train/tensor_act_model_layers_1_self_attn_q_proj/mean": 0.0333709716796875, "train/tensor_act_model_layers_1_self_attn_q_proj/std": 0.7934583210567848, "train/tensor_act_model_layers_1_self_attn_q_proj/max_abs": 5.5, "train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_k_proj/norm": 7514.913359677975, "train/tensor_act_model_layers_1_self_attn_k_proj/mean": 0.047149658203125, "train/tensor_act_model_layers_1_self_attn_k_proj/std": 0.8195828502245655, "train/tensor_act_model_layers_1_self_attn_k_proj/max_abs": 4.84375, "train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_v_proj/norm": 2493.6288743965956, "train/tensor_act_model_layers_1_self_attn_v_proj/mean": 0.0036802291870117188, "train/tensor_act_model_layers_1_self_attn_v_proj/std": 0.271851848590517, "train/tensor_act_model_layers_1_self_attn_v_proj/max_abs": 1.40625, "train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_o_proj/norm": 301.4580922175665, "train/tensor_act_model_layers_1_self_attn_o_proj/mean": -0.0022516250610351562, "train/tensor_act_model_layers_1_self_attn_o_proj/std": 0.03286092571977793, "train/tensor_act_model_layers_1_self_attn_o_proj/max_abs": 0.35546875, "train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn/norm": 301.4580922175665, "train/tensor_act_model_layers_1_self_attn/mean": -0.0022516250610351562, "train/tensor_act_model_layers_1_self_attn/std": 0.03286092571977793, "train/tensor_act_model_layers_1_self_attn/max_abs": 0.35546875, "train/tensor_act_model_layers_1_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_post_attention_layernorm/norm": 9158.569824228702, "train/tensor_act_model_layers_1_post_attention_layernorm/mean": -0.01399993896484375, "train/tensor_act_model_layers_1_post_attention_layernorm/std": 1.0000054609987319, "train/tensor_act_model_layers_1_post_attention_layernorm/max_abs": 4.96875, "train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_mlp_up_proj/norm": 3503.9026158574925, "train/tensor_act_model_layers_1_mlp_up_proj/mean": -0.00506591796875, "train/tensor_act_model_layers_1_mlp_up_proj/std": 0.2207643275831124, "train/tensor_act_model_layers_1_mlp_up_proj/max_abs": 1.2265625, "train/tensor_act_model_layers_1_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_mlp_down_proj/norm": 1282.0529812759387, "train/tensor_act_model_layers_1_mlp_down_proj/mean": 0.0067119598388671875, "train/tensor_act_model_layers_1_mlp_down_proj/std": 0.13964852814435932, "train/tensor_act_model_layers_1_mlp_down_proj/max_abs": 0.88671875, "train/tensor_act_model_layers_1_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_mlp/norm": 1282.0529812759387, "train/tensor_act_model_layers_1_mlp/mean": 0.0067119598388671875, "train/tensor_act_model_layers_1_mlp/std": 0.13964852814435932, "train/tensor_act_model_layers_1_mlp/max_abs": 0.88671875, "train/tensor_act_model_layers_1_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1/norm": 1821.4157893504037, "train/tensor_act_model_layers_1/mean": 0.004604339599609375, "train/tensor_act_model_layers_1/std": 0.1989142884855857, "train/tensor_act_model_layers_1/max_abs": 1.1015625, "train/tensor_act_model_layers_1/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_input_layernorm/norm": 9158.78973389141, "train/tensor_act_model_layers_2_input_layernorm/mean": 0.02327728271484375, "train/tensor_act_model_layers_2_input_layernorm/std": 1.0000010241923734, "train/tensor_act_model_layers_2_input_layernorm/max_abs": 4.46875, "train/tensor_act_model_layers_2_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_q_proj/norm": 9736.292863798752, "train/tensor_act_model_layers_2_self_attn_q_proj/mean": 0.016788482666015625, "train/tensor_act_model_layers_2_self_attn_q_proj/std": 1.0625039114311987, "train/tensor_act_model_layers_2_self_attn_q_proj/max_abs": 5.25, "train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_k_proj/norm": 7735.051202050646, "train/tensor_act_model_layers_2_self_attn_k_proj/mean": 0.015628814697265625, "train/tensor_act_model_layers_2_self_attn_k_proj/std": 0.8444851065697362, "train/tensor_act_model_layers_2_self_attn_k_proj/max_abs": 4.75, "train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_v_proj/norm": 2736.5449248623486, "train/tensor_act_model_layers_2_self_attn_v_proj/mean": -0.0029439926147460938, "train/tensor_act_model_layers_2_self_attn_v_proj/std": 0.29882832015005995, "train/tensor_act_model_layers_2_self_attn_v_proj/max_abs": 1.8671875, "train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_o_proj/norm": 408.9549750343595, "train/tensor_act_model_layers_2_self_attn_o_proj/mean": 0.0020008087158203125, "train/tensor_act_model_layers_2_self_attn_o_proj/std": 0.0446176132967603, "train/tensor_act_model_layers_2_self_attn_o_proj/max_abs": 0.578125, "train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn/norm": 408.9549750343595, "train/tensor_act_model_layers_2_self_attn/mean": 0.0020008087158203125, "train/tensor_act_model_layers_2_self_attn/std": 0.0446176132967603, "train/tensor_act_model_layers_2_self_attn/max_abs": 0.578125, "train/tensor_act_model_layers_2_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_post_attention_layernorm/norm": 9158.794067390949, "train/tensor_act_model_layers_2_post_attention_layernorm/mean": 0.0318145751953125, "train/tensor_act_model_layers_2_post_attention_layernorm/std": 1.0000009535574466, "train/tensor_act_model_layers_2_post_attention_layernorm/max_abs": 4.59375, "train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_mlp_up_proj/norm": 3125.8167549285604, "train/tensor_act_model_layers_2_mlp_up_proj/mean": 2.9494520276784897e-05, "train/tensor_act_model_layers_2_mlp_up_proj/std": 0.1972049672413496, "train/tensor_act_model_layers_2_mlp_up_proj/max_abs": 1.1484375, "train/tensor_act_model_layers_2_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_mlp_down_proj/norm": 1050.6969802776157, "train/tensor_act_model_layers_2_mlp_down_proj/mean": 0.005405426025390625, "train/tensor_act_model_layers_2_mlp_down_proj/std": 0.11462433009314828, "train/tensor_act_model_layers_2_mlp_down_proj/max_abs": 0.66796875, "train/tensor_act_model_layers_2_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_mlp/norm": 1050.6969802776157, "train/tensor_act_model_layers_2_mlp/mean": 0.005405426025390625, "train/tensor_act_model_layers_2_mlp/std": 0.11462433009314828, "train/tensor_act_model_layers_2_mlp/max_abs": 0.66796875, "train/tensor_act_model_layers_2_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2/norm": 2229.1012974584337, "train/tensor_act_model_layers_2/mean": 0.012012481689453125, "train/tensor_act_model_layers_2/std": 0.24310367211460596, "train/tensor_act_model_layers_2/max_abs": 1.328125, "train/tensor_act_model_layers_2/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_input_layernorm/norm": 9158.836547869432, "train/tensor_act_model_layers_3_input_layernorm/mean": 0.04978942871093751, "train/tensor_act_model_layers_3_input_layernorm/std": 1.0000012683441275, "train/tensor_act_model_layers_3_input_layernorm/max_abs": 4.65625, "train/tensor_act_model_layers_3_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_q_proj/norm": 10520.92980011775, "train/tensor_act_model_layers_3_self_attn_q_proj/mean": 0.15374755859375, "train/tensor_act_model_layers_3_self_attn_q_proj/std": 1.138189614442459, "train/tensor_act_model_layers_3_self_attn_q_proj/max_abs": 6.21875, "train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_k_proj/norm": 7931.644892515877, "train/tensor_act_model_layers_3_self_attn_k_proj/mean": 0.08953857421875, "train/tensor_act_model_layers_3_self_attn_k_proj/std": 0.8610920463597765, "train/tensor_act_model_layers_3_self_attn_k_proj/max_abs": 4.03125, "train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_v_proj/norm": 3054.7713515068785, "train/tensor_act_model_layers_3_self_attn_v_proj/mean": -0.002701282501220703, "train/tensor_act_model_layers_3_self_attn_v_proj/std": 0.3339849269576031, "train/tensor_act_model_layers_3_self_attn_v_proj/max_abs": 1.6328125, "train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_o_proj/norm": 486.0344744237919, "train/tensor_act_model_layers_3_self_attn_o_proj/mean": -0.00029712915420532227, "train/tensor_act_model_layers_3_self_attn_o_proj/std": 0.05308823489047627, "train/tensor_act_model_layers_3_self_attn_o_proj/max_abs": 0.47265625, "train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn/norm": 486.0344744237919, "train/tensor_act_model_layers_3_self_attn/mean": -0.00029712915420532227, "train/tensor_act_model_layers_3_self_attn/std": 0.05308823489047627, "train/tensor_act_model_layers_3_self_attn/max_abs": 0.47265625, "train/tensor_act_model_layers_3_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_post_attention_layernorm/norm": 9158.835632337961, "train/tensor_act_model_layers_3_post_attention_layernorm/mean": 0.04740905761718751, "train/tensor_act_model_layers_3_post_attention_layernorm/std": 1.0000005670588714, "train/tensor_act_model_layers_3_post_attention_layernorm/max_abs": 4.8125, "train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_mlp_up_proj/norm": 3097.754188140022, "train/tensor_act_model_layers_3_mlp_up_proj/mean": -0.0009462833404541016, "train/tensor_act_model_layers_3_mlp_up_proj/std": 0.19531258106913943, "train/tensor_act_model_layers_3_mlp_up_proj/max_abs": 1.2109375, "train/tensor_act_model_layers_3_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_mlp_down_proj/norm": 998.9542597675202, "train/tensor_act_model_layers_3_mlp_down_proj/mean": -0.0022554397583007812, "train/tensor_act_model_layers_3_mlp_down_proj/std": 0.10897847648259888, "train/tensor_act_model_layers_3_mlp_down_proj/max_abs": 0.66015625, "train/tensor_act_model_layers_3_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_mlp/norm": 998.9542597675202, "train/tensor_act_model_layers_3_mlp/mean": -0.0022554397583007812, "train/tensor_act_model_layers_3_mlp/std": 0.10897847648259888, "train/tensor_act_model_layers_3_mlp/max_abs": 0.66015625, "train/tensor_act_model_layers_3_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3/norm": 2475.276032622091, "train/tensor_act_model_layers_3/mean": 0.00946044921875, "train/tensor_act_model_layers_3/std": 0.2701432434459629, "train/tensor_act_model_layers_3/max_abs": 1.484375, "train/tensor_act_model_layers_3/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_input_layernorm/norm": 9158.859802255498, "train/tensor_act_model_layers_4_input_layernorm/mean": 0.03276824951171874, "train/tensor_act_model_layers_4_input_layernorm/std": 1.0000005615582506, "train/tensor_act_model_layers_4_input_layernorm/max_abs": 4.8125, "train/tensor_act_model_layers_4_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_q_proj/norm": 8444.509770831344, "train/tensor_act_model_layers_4_self_attn_q_proj/mean": 0.062774658203125, "train/tensor_act_model_layers_4_self_attn_q_proj/std": 0.9201686668813257, "train/tensor_act_model_layers_4_self_attn_q_proj/max_abs": 4.78125, "train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_k_proj/norm": 7745.710212624892, "train/tensor_act_model_layers_4_self_attn_k_proj/mean": 0.017276763916015625, "train/tensor_act_model_layers_4_self_attn_k_proj/std": 0.845461853783858, "train/tensor_act_model_layers_4_self_attn_k_proj/max_abs": 4.21875, "train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_v_proj/norm": 3085.8902545445458, "train/tensor_act_model_layers_4_self_attn_v_proj/mean": 0.0008850693702697754, "train/tensor_act_model_layers_4_self_attn_v_proj/std": 0.33703761500531826, "train/tensor_act_model_layers_4_self_attn_v_proj/max_abs": 1.7734375, "train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_o_proj/norm": 580.2747154684458, "train/tensor_act_model_layers_4_self_attn_o_proj/mean": 0.0011477470397949219, "train/tensor_act_model_layers_4_self_attn_o_proj/std": 0.06326528385724953, "train/tensor_act_model_layers_4_self_attn_o_proj/max_abs": 0.60546875, "train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn/norm": 580.2747154684458, "train/tensor_act_model_layers_4_self_attn/mean": 0.0011477470397949219, "train/tensor_act_model_layers_4_self_attn/std": 0.06326528385724953, "train/tensor_act_model_layers_4_self_attn/max_abs": 0.60546875, "train/tensor_act_model_layers_4_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_post_attention_layernorm/norm": 9158.856872565368, "train/tensor_act_model_layers_4_post_attention_layernorm/mean": 0.0382232666015625, "train/tensor_act_model_layers_4_post_attention_layernorm/std": 1.000000227126267, "train/tensor_act_model_layers_4_post_attention_layernorm/max_abs": 5.21875, "train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_mlp_up_proj/norm": 3015.185914880472, "train/tensor_act_model_layers_4_mlp_up_proj/mean": -0.000986814498901367, "train/tensor_act_model_layers_4_mlp_up_proj/std": 0.19018606370318258, "train/tensor_act_model_layers_4_mlp_up_proj/max_abs": 1.1015625, "train/tensor_act_model_layers_4_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_mlp_down_proj/norm": 921.3529462980846, "train/tensor_act_model_layers_4_mlp_down_proj/mean": -0.0026712417602539062, "train/tensor_act_model_layers_4_mlp_down_proj/std": 0.10058596968487546, "train/tensor_act_model_layers_4_mlp_down_proj/max_abs": 0.5390625, "train/tensor_act_model_layers_4_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_mlp/norm": 921.3529462980846, "train/tensor_act_model_layers_4_mlp/mean": -0.0026712417602539062, "train/tensor_act_model_layers_4_mlp/std": 0.10058596968487546, "train/tensor_act_model_layers_4_mlp/max_abs": 0.5390625, "train/tensor_act_model_layers_4_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4/norm": 2564.812437940014, "train/tensor_act_model_layers_4/mean": 0.007947921752929688, "train/tensor_act_model_layers_4/std": 0.2797865139362316, "train/tensor_act_model_layers_4/max_abs": 1.65625, "train/tensor_act_model_layers_4/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_input_layernorm/norm": 9158.861267098808, "train/tensor_act_model_layers_5_input_layernorm/mean": 0.0275726318359375, "train/tensor_act_model_layers_5_input_layernorm/std": 1.0000002960441194, "train/tensor_act_model_layers_5_input_layernorm/max_abs": 4.84375, "train/tensor_act_model_layers_5_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_q_proj/norm": 8626.440264451365, "train/tensor_act_model_layers_5_self_attn_q_proj/mean": -0.0014693737030029297, "train/tensor_act_model_layers_5_self_attn_q_proj/std": 0.9414076213487585, "train/tensor_act_model_layers_5_self_attn_q_proj/max_abs": 6.0, "train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_k_proj/norm": 8086.794526499998, "train/tensor_act_model_layers_5_self_attn_k_proj/mean": 0.014064788818359375, "train/tensor_act_model_layers_5_self_attn_k_proj/std": 0.8823257181255008, "train/tensor_act_model_layers_5_self_attn_k_proj/max_abs": 5.15625, "train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_v_proj/norm": 2959.718677812412, "train/tensor_act_model_layers_5_self_attn_v_proj/mean": -0.003086090087890625, "train/tensor_act_model_layers_5_self_attn_v_proj/std": 0.32312162517924214, "train/tensor_act_model_layers_5_self_attn_v_proj/max_abs": 1.84375, "train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_o_proj/norm": 497.6350976881482, "train/tensor_act_model_layers_5_self_attn_o_proj/mean": -0.0008254051208496094, "train/tensor_act_model_layers_5_self_attn_o_proj/std": 0.05436818972384846, "train/tensor_act_model_layers_5_self_attn_o_proj/max_abs": 0.4375, "train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn/norm": 497.6350976881482, "train/tensor_act_model_layers_5_self_attn/mean": -0.0008254051208496094, "train/tensor_act_model_layers_5_self_attn/std": 0.05436818972384846, "train/tensor_act_model_layers_5_self_attn/max_abs": 0.4375, "train/tensor_act_model_layers_5_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_post_attention_layernorm/norm": 9158.866821301317, "train/tensor_act_model_layers_5_post_attention_layernorm/mean": 0.0240020751953125, "train/tensor_act_model_layers_5_post_attention_layernorm/std": 1.00000024016478, "train/tensor_act_model_layers_5_post_attention_layernorm/max_abs": 4.71875, "train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_mlp_up_proj/norm": 3031.677025734979, "train/tensor_act_model_layers_5_mlp_up_proj/mean": -0.0006145238876342773, "train/tensor_act_model_layers_5_mlp_up_proj/std": 0.19134540618480564, "train/tensor_act_model_layers_5_mlp_up_proj/max_abs": 1.1171875, "train/tensor_act_model_layers_5_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_mlp_down_proj/norm": 929.9160394931031, "train/tensor_act_model_layers_5_mlp_down_proj/mean": 0.0011644363403320312, "train/tensor_act_model_layers_5_mlp_down_proj/std": 0.1015626140108337, "train/tensor_act_model_layers_5_mlp_down_proj/max_abs": 0.55859375, "train/tensor_act_model_layers_5_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_mlp/norm": 929.9160394931031, "train/tensor_act_model_layers_5_mlp/mean": 0.0011644363403320312, "train/tensor_act_model_layers_5_mlp/std": 0.1015626140108337, "train/tensor_act_model_layers_5_mlp/max_abs": 0.55859375, "train/tensor_act_model_layers_5_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5/norm": 2580.3944732393816, "train/tensor_act_model_layers_5/mean": 0.008279800415039062, "train/tensor_act_model_layers_5/std": 0.2813725806040801, "train/tensor_act_model_layers_5/max_abs": 1.5, "train/tensor_act_model_layers_5/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_input_layernorm/norm": 9158.862487805487, "train/tensor_act_model_layers_6_input_layernorm/mean": 0.02984619140625, "train/tensor_act_model_layers_6_input_layernorm/std": 1.0000004451720916, "train/tensor_act_model_layers_6_input_layernorm/max_abs": 4.65625, "train/tensor_act_model_layers_6_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_q_proj/norm": 10739.079023412945, "train/tensor_act_model_layers_6_self_attn_q_proj/mean": -0.05267333984375, "train/tensor_act_model_layers_6_self_attn_q_proj/std": 1.172365214812159, "train/tensor_act_model_layers_6_self_attn_q_proj/max_abs": 6.03125, "train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_k_proj/norm": 9821.311569001531, "train/tensor_act_model_layers_6_self_attn_k_proj/mean": -0.085968017578125, "train/tensor_act_model_layers_6_self_attn_k_proj/std": 1.068372004136195, "train/tensor_act_model_layers_6_self_attn_k_proj/max_abs": 4.40625, "train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_v_proj/norm": 3351.5042061824015, "train/tensor_act_model_layers_6_self_attn_v_proj/mean": -0.0036039352416992188, "train/tensor_act_model_layers_6_self_attn_v_proj/std": 0.36547945254426323, "train/tensor_act_model_layers_6_self_attn_v_proj/max_abs": 2.328125, "train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_o_proj/norm": 798.792740246835, "train/tensor_act_model_layers_6_self_attn_o_proj/mean": -0.002144336700439453, "train/tensor_act_model_layers_6_self_attn_o_proj/std": 0.08719287717656642, "train/tensor_act_model_layers_6_self_attn_o_proj/max_abs": 1.046875, "train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn/norm": 798.792740246835, "train/tensor_act_model_layers_6_self_attn/mean": -0.002144336700439453, "train/tensor_act_model_layers_6_self_attn/std": 0.08719287717656642, "train/tensor_act_model_layers_6_self_attn/max_abs": 1.046875, "train/tensor_act_model_layers_6_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_post_attention_layernorm/norm": 9158.868591321168, "train/tensor_act_model_layers_6_post_attention_layernorm/mean": 0.02191925048828125, "train/tensor_act_model_layers_6_post_attention_layernorm/std": 1.000000366911923, "train/tensor_act_model_layers_6_post_attention_layernorm/max_abs": 4.875, "train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_mlp_up_proj/norm": 3078.6657922183363, "train/tensor_act_model_layers_6_mlp_up_proj/mean": 0.0003501623868942261, "train/tensor_act_model_layers_6_mlp_up_proj/std": 0.19427530100764198, "train/tensor_act_model_layers_6_mlp_up_proj/max_abs": 1.0859375, "train/tensor_act_model_layers_6_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_mlp_down_proj/norm": 1012.1440503813626, "train/tensor_act_model_layers_6_mlp_down_proj/mean": 0.0057201385498046875, "train/tensor_act_model_layers_6_mlp_down_proj/std": 0.11035170806644656, "train/tensor_act_model_layers_6_mlp_down_proj/max_abs": 0.6171875, "train/tensor_act_model_layers_6_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_mlp/norm": 1012.1440503813626, "train/tensor_act_model_layers_6_mlp/mean": 0.0057201385498046875, "train/tensor_act_model_layers_6_mlp/std": 0.11035170806644656, "train/tensor_act_model_layers_6_mlp/max_abs": 0.6171875, "train/tensor_act_model_layers_6_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6/norm": 2675.436665474263, "train/tensor_act_model_layers_6/mean": 0.011852264404296877, "train/tensor_act_model_layers_6/std": 0.2915051843774514, "train/tensor_act_model_layers_6/max_abs": 1.6640625, "train/tensor_act_model_layers_6/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_input_layernorm/norm": 9158.866333015456, "train/tensor_act_model_layers_7_input_layernorm/mean": 0.04217529296875, "train/tensor_act_model_layers_7_input_layernorm/std": 1.0000001769512734, "train/tensor_act_model_layers_7_input_layernorm/max_abs": 4.78125, "train/tensor_act_model_layers_7_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_q_proj/norm": 10461.691153606722, "train/tensor_act_model_layers_7_self_attn_q_proj/mean": -0.0406341552734375, "train/tensor_act_model_layers_7_self_attn_q_proj/std": 1.1416046069727064, "train/tensor_act_model_layers_7_self_attn_q_proj/max_abs": 7.0, "train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_k_proj/norm": 10485.176929176645, "train/tensor_act_model_layers_7_self_attn_k_proj/mean": -0.0906982421875, "train/tensor_act_model_layers_7_self_attn_k_proj/std": 1.1411216973846043, "train/tensor_act_model_layers_7_self_attn_k_proj/max_abs": 5.21875, "train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_v_proj/norm": 3279.989591989278, "train/tensor_act_model_layers_7_self_attn_v_proj/mean": 0.01746368408203125, "train/tensor_act_model_layers_7_self_attn_v_proj/std": 0.35742205340675426, "train/tensor_act_model_layers_7_self_attn_v_proj/max_abs": 2.046875, "train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_o_proj/norm": 870.6736716013772, "train/tensor_act_model_layers_7_self_attn_o_proj/mean": -0.0035886764526367188, "train/tensor_act_model_layers_7_self_attn_o_proj/std": 0.0950324238613928, "train/tensor_act_model_layers_7_self_attn_o_proj/max_abs": 0.8203125, "train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn/norm": 870.6736716013772, "train/tensor_act_model_layers_7_self_attn/mean": -0.0035886764526367188, "train/tensor_act_model_layers_7_self_attn/std": 0.0950324238613928, "train/tensor_act_model_layers_7_self_attn/max_abs": 0.8203125, "train/tensor_act_model_layers_7_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_post_attention_layernorm/norm": 9158.870239264383, "train/tensor_act_model_layers_7_post_attention_layernorm/mean": 0.02803802490234375, "train/tensor_act_model_layers_7_post_attention_layernorm/std": 1.0000002372252939, "train/tensor_act_model_layers_7_post_attention_layernorm/max_abs": 4.6875, "train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_mlp_up_proj/norm": 3136.1940636118106, "train/tensor_act_model_layers_7_mlp_up_proj/mean": -0.0019998550415039067, "train/tensor_act_model_layers_7_mlp_up_proj/std": 0.1975713855186142, "train/tensor_act_model_layers_7_mlp_up_proj/max_abs": 1.1953125, "train/tensor_act_model_layers_7_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_mlp_down_proj/norm": 1157.7220554563974, "train/tensor_act_model_layers_7_mlp_down_proj/mean": -0.0019550323486328125, "train/tensor_act_model_layers_7_mlp_down_proj/std": 0.12609909852702159, "train/tensor_act_model_layers_7_mlp_down_proj/max_abs": 0.73046875, "train/tensor_act_model_layers_7_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_mlp/norm": 1157.7220554563974, "train/tensor_act_model_layers_7_mlp/mean": -0.0019550323486328125, "train/tensor_act_model_layers_7_mlp/std": 0.12609909852702159, "train/tensor_act_model_layers_7_mlp/max_abs": 0.73046875, "train/tensor_act_model_layers_7_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7/norm": 2730.395501469702, "train/tensor_act_model_layers_7/mean": 0.0063190460205078125, "train/tensor_act_model_layers_7/std": 0.29797527457897055, "train/tensor_act_model_layers_7/max_abs": 1.640625, "train/tensor_act_model_layers_7/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_input_layernorm/norm": 9158.865478523945, "train/tensor_act_model_layers_8_input_layernorm/mean": 0.021881103515625003, "train/tensor_act_model_layers_8_input_layernorm/std": 1.000000241678179, "train/tensor_act_model_layers_8_input_layernorm/max_abs": 4.53125, "train/tensor_act_model_layers_8_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_q_proj/norm": 12639.429603678956, "train/tensor_act_model_layers_8_self_attn_q_proj/mean": -0.048736572265625, "train/tensor_act_model_layers_8_self_attn_q_proj/std": 1.3794029002653125, "train/tensor_act_model_layers_8_self_attn_q_proj/max_abs": 6.09375, "train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_k_proj/norm": 11854.747864460993, "train/tensor_act_model_layers_8_self_attn_k_proj/mean": -0.0522308349609375, "train/tensor_act_model_layers_8_self_attn_k_proj/std": 1.293951040201286, "train/tensor_act_model_layers_8_self_attn_k_proj/max_abs": 6.375, "train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_v_proj/norm": 3463.3056672450866, "train/tensor_act_model_layers_8_self_attn_v_proj/mean": -0.0057849884033203125, "train/tensor_act_model_layers_8_self_attn_v_proj/std": 0.37829724816102006, "train/tensor_act_model_layers_8_self_attn_v_proj/max_abs": 1.921875, "train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_o_proj/norm": 1078.046025070771, "train/tensor_act_model_layers_8_self_attn_o_proj/mean": 0.0038623809814453125, "train/tensor_act_model_layers_8_self_attn_o_proj/std": 0.11758523080684756, "train/tensor_act_model_layers_8_self_attn_o_proj/max_abs": 0.8125, "train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn/norm": 1078.046025070771, "train/tensor_act_model_layers_8_self_attn/mean": 0.0038623809814453125, "train/tensor_act_model_layers_8_self_attn/std": 0.11758523080684756, "train/tensor_act_model_layers_8_self_attn/max_abs": 0.8125, "train/tensor_act_model_layers_8_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_post_attention_layernorm/norm": 9158.873291019005, "train/tensor_act_model_layers_8_post_attention_layernorm/mean": 0.03253173828125, "train/tensor_act_model_layers_8_post_attention_layernorm/std": 1.0000002402811954, "train/tensor_act_model_layers_8_post_attention_layernorm/max_abs": 4.71875, "train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_mlp_up_proj/norm": 3753.7269197260766, "train/tensor_act_model_layers_8_mlp_up_proj/mean": 0.00576019287109375, "train/tensor_act_model_layers_8_mlp_up_proj/std": 0.23651187933555926, "train/tensor_act_model_layers_8_mlp_up_proj/max_abs": 1.3671875, "train/tensor_act_model_layers_8_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_mlp_down_proj/norm": 1833.4078311255546, "train/tensor_act_model_layers_8_mlp_down_proj/mean": -0.0014677047729492188, "train/tensor_act_model_layers_8_mlp_down_proj/std": 0.20031891610751495, "train/tensor_act_model_layers_8_mlp_down_proj/max_abs": 1.1640625, "train/tensor_act_model_layers_8_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_mlp/norm": 1833.4078311255546, "train/tensor_act_model_layers_8_mlp/mean": -0.0014677047729492188, "train/tensor_act_model_layers_8_mlp/std": 0.20031891610751495, "train/tensor_act_model_layers_8_mlp/max_abs": 1.1640625, "train/tensor_act_model_layers_8_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8/norm": 3714.703350497436, "train/tensor_act_model_layers_8/mean": 0.008716583251953125, "train/tensor_act_model_layers_8/std": 0.40576329341056694, "train/tensor_act_model_layers_8/max_abs": 2.03125, "train/tensor_act_model_layers_8/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8/frac_near_user_limit": 0.0, "train/tensor_act_model_norm/norm": 9158.885131843323, "train/tensor_act_model_norm/mean": 0.0232391357421875, "train/tensor_act_model_norm/std": 1.0000002746237064, "train/tensor_act_model_norm/max_abs": 4.75, "train/tensor_act_model_norm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_norm/frac_near_user_limit": 0.0, "train/tensor_act_model/norm": 9158.885131843323, "train/tensor_act_model/mean": 0.0232391357421875, "train/tensor_act_model/std": 1.0000002746237064, "train/tensor_act_model/max_abs": 4.75, "train/tensor_act_model/frac_near_dtype_limit": 0.0, "train/tensor_act_model/frac_near_user_limit": 0.0, "train/tensor_act_lm_head/norm": 137978.5112112536, "train/tensor_act_lm_head/mean": -2.0576171875000004, "train/tensor_act_lm_head/std": 1.6914375425620019, "train/tensor_act_lm_head/max_abs": 11.8125, "train/tensor_act_lm_head/frac_near_dtype_limit": 0.0, "train/tensor_act_lm_head/frac_near_user_limit": 0.0, "train/tensor_act_/norm": 12.34277911184733, "train/tensor_act_/mean": 3.085545063018799, "train/tensor_act_/std": 0.0, "train/tensor_act_/max_abs": 3.1596059799194336, "train/tensor_act_/frac_near_dtype_limit": 0.0, "train/tensor_act_/frac_near_user_limit": 0.0, "train/tensor_grad_model_norm_weight/norm": 0.3140373561530066, "train/tensor_grad_model_norm_weight/mean": -0.006618499755859375, "train/tensor_grad_model_norm_weight/std": 0.002068276652680219, "train/tensor_grad_model_norm_weight/max_abs": 0.01385498046875, "train/tensor_grad_model_norm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_norm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/norm": 0.26025230307022634, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/mean": -2.4600012693554163e-08, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/std": 0.00029343398646999646, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/max_abs": 0.002655029296875, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/norm": 0.4110867073502657, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/mean": -4.274479579180479e-07, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/std": 0.00046367620843929094, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/max_abs": 0.002716064453125, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/norm": 0.019447025851964624, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/mean": 3.9157457649707794e-05, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/std": 0.000429591868268785, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/max_abs": 0.0015716552734375, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/norm": 0.30612391776426706, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/mean": -4.833564162254333e-07, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/std": 0.0005979014626018539, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/max_abs": 0.004791259765625, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/norm": 0.7119194208474616, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/mean": 4.765111953020096e-06, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/std": 0.0013908427641118966, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/max_abs": 0.009765625, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/norm": 0.16623753694556406, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/mean": 1.1397423804737628e-06, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/std": 0.000324684607977906, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/max_abs": 0.0038604736328125, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/norm": 0.10740945917684312, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/mean": -1.043081283569336e-06, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/std": 0.00020978625517021878, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/max_abs": 0.0018463134765625, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_input_layernorm_weight/norm": 0.028055232988969746, "train/tensor_grad_model_layers_8_input_layernorm_weight/mean": -0.00017432868480682373, "train/tensor_grad_model_layers_8_input_layernorm_weight/std": 0.0005966479081292068, "train/tensor_grad_model_layers_8_input_layernorm_weight/max_abs": 0.0020599365234375, "train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/norm": 0.4239274009773825, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/mean": 6.979462341405451e-07, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/std": 0.00047807733058092964, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/max_abs": 0.0040283203125, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/norm": 0.5500951511265744, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/mean": 8.427305147051811e-07, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/std": 0.0006207572805987063, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/max_abs": 0.00592041015625, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/norm": 0.018391493845650374, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/mean": -1.1431053280830383e-05, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/std": 0.0004077178083436662, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/max_abs": 0.0019989013671875, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/norm": 0.2792116048333057, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/mean": -1.1397060006856918e-06, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/std": 0.0005453423260938054, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/max_abs": 0.004913330078125, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/norm": 0.5816282374777465, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/mean": -3.025401383638382e-06, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/std": 0.0011359983194955572, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/max_abs": 0.0093994140625, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/norm": 0.2207897995078383, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/mean": 1.3531680451706052e-06, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/std": 0.0004312305883395487, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/max_abs": 0.006072998046875, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/norm": 0.11642218871094213, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/mean": -2.4102628231048584e-06, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/std": 0.0002273880887448863, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/max_abs": 0.00244140625, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_input_layernorm_weight/norm": 0.02808577331898204, "train/tensor_grad_model_layers_7_input_layernorm_weight/mean": 8.419447112828502e-07, "train/tensor_grad_model_layers_7_input_layernorm_weight/std": 0.000623451731167683, "train/tensor_grad_model_layers_7_input_layernorm_weight/max_abs": 0.0028076171875, "train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/norm": 0.5091221281177153, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/mean": 1.9668368622660633e-06, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/std": 0.0005738216679627089, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/max_abs": 0.005035400390625, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/norm": 0.6348685313220278, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/mean": 2.6694033294916153e-07, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/std": 0.0007164730821052101, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/max_abs": 0.006103515625, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/norm": 0.020273390128607707, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/mean": 5.274266004562378e-05, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/std": 0.0004466476153536851, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/max_abs": 0.0020751953125, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/norm": 0.330183565352083, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/mean": -1.8938444554805756e-06, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/std": 0.0006448974090784716, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/max_abs": 0.004425048828125, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/norm": 0.6757650208040404, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/mean": 8.448958396911621e-06, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/std": 0.0013198556532964516, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/max_abs": 0.0098876953125, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/norm": 0.1486808388473866, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/mean": -1.889653503894806e-06, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/std": 0.0002903939882593242, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/max_abs": 0.003509521484375, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/norm": 0.08525953765832373, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/mean": -2.3588654585182667e-07, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/std": 0.00016652373622645598, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/max_abs": 0.00115203857421875, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_input_layernorm_weight/norm": 0.03215523252833009, "train/tensor_grad_model_layers_6_input_layernorm_weight/mean": -0.0001233704388141632, "train/tensor_grad_model_layers_6_input_layernorm_weight/std": 0.0007029179815078741, "train/tensor_grad_model_layers_6_input_layernorm_weight/max_abs": 0.0034332275390625, "train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/norm": 0.5447748457558951, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/mean": -6.993068382143974e-07, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/std": 0.000614104388010392, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/max_abs": 0.004608154296875, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/norm": 0.7220449800719276, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/mean": 1.3120006769895554e-07, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/std": 0.0008139678337106606, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/max_abs": 0.005706787109375, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/norm": 0.018823162034157223, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/mean": 4.1157007217407227e-05, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/std": 0.00041555377899105605, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/max_abs": 0.00140380859375, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/norm": 0.3505199657200831, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/mean": 1.3037933968007565e-06, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/std": 0.0006846107163238569, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/max_abs": 0.007171630859375, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/norm": 0.520757618804886, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/mean": 1.1994852684438229e-06, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/std": 0.001017107616730838, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/max_abs": 0.01092529296875, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/norm": 0.09609577828104279, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/mean": 1.6889534890651703e-06, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/std": 0.00018768759095130736, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/max_abs": 0.0019989013671875, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/norm": 0.05909136233456057, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/mean": 5.144684109836817e-07, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/std": 0.00011541363059973877, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/max_abs": 0.00162506103515625, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_input_layernorm_weight/norm": 0.01664093901557326, "train/tensor_grad_model_layers_5_input_layernorm_weight/mean": 6.784126162528992e-05, "train/tensor_grad_model_layers_5_input_layernorm_weight/std": 0.00036285050479602693, "train/tensor_grad_model_layers_5_input_layernorm_weight/max_abs": 0.001708984375, "train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/norm": 0.569249665150902, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/mean": 2.1673622541129593e-06, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/std": 0.0006417847780511843, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/max_abs": 0.00494384765625, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/norm": 0.6969099232771115, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/mean": 3.027264028787613e-06, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/std": 0.0007855715454180655, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/max_abs": 0.00537109375, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/norm": 0.01828787353542204, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/mean": -7.230043411254883e-05, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/std": 0.0003992499171670238, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/max_abs": 0.0018768310546875, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/norm": 0.32410548426276165, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/mean": -1.3430617400445044e-06, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/std": 0.0006330197896350134, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/max_abs": 0.005279541015625, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/norm": 0.38665123552465125, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/mean": -1.292908564209938e-06, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/std": 0.0007551836297561284, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/max_abs": 0.00689697265625, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/norm": 0.07685140577189588, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/mean": -1.5028781490400434e-06, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/std": 0.00015010444818410604, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/max_abs": 0.001800537109375, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/norm": 0.06249469496146752, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/mean": 8.378628990612924e-07, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/std": 0.00012209175806172248, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/max_abs": 0.00142669677734375, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_input_layernorm_weight/norm": 0.014499166008893285, "train/tensor_grad_model_layers_4_input_layernorm_weight/mean": -0.00012033060193061829, "train/tensor_grad_model_layers_4_input_layernorm_weight/std": 0.0002978021794515658, "train/tensor_grad_model_layers_4_input_layernorm_weight/max_abs": 0.001190185546875, "train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/norm": 0.6348835527475227, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/mean": 2.726039383560419e-06, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/std": 0.0007156685057058643, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/max_abs": 0.00628662109375, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/norm": 0.7933408869322305, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/mean": 3.46451997756958e-06, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/std": 0.0008954833779846271, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/max_abs": 0.00823974609375, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/norm": 0.02216423019948128, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/mean": -8.822977542877197e-05, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/std": 0.0004835319855374909, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/max_abs": 0.002593994140625, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/norm": 0.30735278195334903, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/mean": -5.939946277067065e-07, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/std": 0.0006003041614608425, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/max_abs": 0.00531005859375, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/norm": 0.47152372510489016, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/mean": -1.801163307391107e-06, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/std": 0.0009209493170190383, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/max_abs": 0.007171630859375, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/norm": 0.11293648181112204, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/mean": -1.7615966498851776e-06, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/std": 0.00022058114423191606, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/max_abs": 0.0022735595703125, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/norm": 0.059437744968443, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/mean": 1.8550781533122063e-07, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/std": 0.00011609377149315461, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/max_abs": 0.00091552734375, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_input_layernorm_weight/norm": 0.016322047857146693, "train/tensor_grad_model_layers_3_input_layernorm_weight/mean": -2.587307244539261e-05, "train/tensor_grad_model_layers_3_input_layernorm_weight/std": 0.0003610962964440077, "train/tensor_grad_model_layers_3_input_layernorm_weight/max_abs": 0.001678466796875, "train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/norm": 0.7576419599101569, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/mean": -2.082379069179297e-06, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/std": 0.0008548833918412956, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/max_abs": 0.007598876953125, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/norm": 0.9297782789516292, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/mean": 6.190501153469086e-06, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/std": 0.0010473705549273293, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/max_abs": 0.0074462890625, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/norm": 0.02594651082943509, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/mean": -7.686018943786621e-05, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/std": 0.0005703787486938851, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/max_abs": 0.00193023681640625, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/norm": 0.32488650761207233, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/mean": 4.579778760671616e-06, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/std": 0.000634546290983288, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/max_abs": 0.005889892578125, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/norm": 0.5176637974496008, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/mean": 6.784684956073761e-06, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/std": 0.001011064996978713, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/max_abs": 0.00714111328125, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/norm": 0.10246288558032977, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/mean": -3.6008714232593775e-07, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/std": 0.00020012373680981202, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/max_abs": 0.0022430419921875, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/norm": 0.0647402350765535, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/mean": 7.437483873218298e-07, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/std": 0.00012644644580173545, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/max_abs": 0.001495361328125, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_input_layernorm_weight/norm": 0.017496555534344645, "train/tensor_grad_model_layers_2_input_layernorm_weight/mean": 0.00012053549289703369, "train/tensor_grad_model_layers_2_input_layernorm_weight/std": 0.00036869786499644087, "train/tensor_grad_model_layers_2_input_layernorm_weight/max_abs": 0.0019378662109375, "train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/norm": 1.0471109152906861, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/mean": 4.7208741307258614e-06, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/std": 0.0011805190822957745, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/max_abs": 0.01129150390625, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/norm": 1.1280915907881908, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/mean": 1.3246608432382343e-06, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/std": 0.0012711289102917448, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/max_abs": 0.00927734375, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/norm": 0.03569999716599103, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/mean": -5.647540092468262e-05, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/std": 0.0007896023136567698, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/max_abs": 0.00347900390625, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/norm": 0.43539001650984194, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/mean": 9.60018951445818e-07, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/std": 0.0008503758163232629, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/max_abs": 0.0111083984375, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/norm": 0.6962897473261009, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/mean": 1.1696829460561277e-07, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/std": 0.0013599450739407621, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/max_abs": 0.01153564453125, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/norm": 0.0906099734509409, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/mean": 8.337037797900848e-07, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/std": 0.00017697632508598447, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/max_abs": 0.002410888671875, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/norm": 0.0728063202556278, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/mean": 1.557054929435253e-08, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/std": 0.00014220218988788194, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/max_abs": 0.0015869140625, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_input_layernorm_weight/norm": 0.018508678412802338, "train/tensor_grad_model_layers_1_input_layernorm_weight/mean": 2.3896805942058563e-05, "train/tensor_grad_model_layers_1_input_layernorm_weight/std": 0.0004098031937883901, "train/tensor_grad_model_layers_1_input_layernorm_weight/max_abs": 0.0017547607421875, "train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/norm": 1.4804172154618904, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/mean": -4.631292540580035e-06, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/std": 0.0016695101003668653, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/max_abs": 0.01483154296875, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/norm": 2.206872228784369, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/mean": 7.352093234658241e-06, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/std": 0.002488334277224233, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/max_abs": 0.015380859375, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/norm": 0.08442042505521935, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/mean": 5.408190190792084e-05, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/std": 0.0018718482317452295, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/max_abs": 0.007232666015625, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/norm": 1.0784712525676086, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/mean": 6.760121323168278e-06, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/std": 0.0021064004423660735, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/max_abs": 0.027099609375, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/norm": 1.7430773459698234, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/mean": -7.91158527135849e-06, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/std": 0.0034055304413431733, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/max_abs": 0.0286865234375, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/norm": 0.07944339313752548, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/mean": 1.2395321391522884e-07, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/std": 0.0001551661666112421, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/max_abs": 0.00147247314453125, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/norm": 0.057045541566442495, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/mean": 4.0802115108817816e-07, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/std": 0.00011141785897713587, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/max_abs": 0.001007080078125, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_input_layernorm_weight/norm": 0.044414914575084743, "train/tensor_grad_model_layers_0_input_layernorm_weight/mean": 0.00012800097465515137, "train/tensor_grad_model_layers_0_input_layernorm_weight/std": 0.0009775568531857459, "train/tensor_grad_model_layers_0_input_layernorm_weight/max_abs": 0.00445556640625, "train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_embed_tokens_weight/norm": 1.3963682753312472, "train/tensor_grad_model_embed_tokens_weight/mean": -6.477348506450653e-07, "train/tensor_grad_model_embed_tokens_weight/std": 0.00048201806107111585, "train/tensor_grad_model_embed_tokens_weight/max_abs": 0.05859375, "train/tensor_grad_model_embed_tokens_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_embed_tokens_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_embed_tokens_weight/norm": 47.75, "train/tensor_param_model_embed_tokens_weight/mean": -0.00093841552734375, "train/tensor_param_model_embed_tokens_weight/std": 0.06591796875, "train/tensor_param_model_embed_tokens_weight/max_abs": 0.25, "train/tensor_param_model_embed_tokens_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_embed_tokens_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/norm": 4.625, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/mean": -4.673004150390625e-05, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/std": 0.0361328125, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/max_abs": 0.138671875, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/norm": 4.21875, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/mean": 0.0001049041748046875, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/std": 0.032958984375, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/max_abs": 0.1328125, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/mean": 0.0001068115234375, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/max_abs": 0.08251953125, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/norm": 2.546875, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/mean": 1.424551010131836e-05, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/std": 0.0198974609375, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/max_abs": 0.08544921875, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_mlp_up_proj_weight/norm": 4.6875, "train/tensor_param_model_layers_0_mlp_up_proj_weight/mean": -7.200241088867188e-05, "train/tensor_param_model_layers_0_mlp_up_proj_weight/std": 0.0211181640625, "train/tensor_param_model_layers_0_mlp_up_proj_weight/max_abs": 0.0859375, "train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_mlp_down_proj_weight/norm": 4.75, "train/tensor_param_model_layers_0_mlp_down_proj_weight/mean": 7.05718994140625e-05, "train/tensor_param_model_layers_0_mlp_down_proj_weight/std": 0.0213623046875, "train/tensor_param_model_layers_0_mlp_down_proj_weight/max_abs": 0.09033203125, "train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_0_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_0_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_0_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/norm": 4.78125, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/mean": -9.870529174804688e-05, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/std": 0.037353515625, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/max_abs": 0.1494140625, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/norm": 4.84375, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/mean": 0.00026702880859375, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/std": 0.037841796875, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/max_abs": 0.166015625, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/norm": 2.796875, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/mean": -0.00028228759765625, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/std": 0.0218505859375, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/max_abs": 0.080078125, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/norm": 2.84375, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/mean": -4.38690185546875e-05, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/std": 0.022216796875, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/max_abs": 0.076171875, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_mlp_up_proj_weight/norm": 4.59375, "train/tensor_param_model_layers_1_mlp_up_proj_weight/mean": -5.602836608886719e-05, "train/tensor_param_model_layers_1_mlp_up_proj_weight/std": 0.020751953125, "train/tensor_param_model_layers_1_mlp_up_proj_weight/max_abs": 0.08154296875, "train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_mlp_down_proj_weight/norm": 4.59375, "train/tensor_param_model_layers_1_mlp_down_proj_weight/mean": -2.944469451904297e-05, "train/tensor_param_model_layers_1_mlp_down_proj_weight/std": 0.020751953125, "train/tensor_param_model_layers_1_mlp_down_proj_weight/max_abs": 0.080078125, "train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_1_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_1_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_1_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/norm": 5.21875, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/mean": -0.000484466552734375, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/std": 0.040771484375, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/max_abs": 0.2314453125, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/norm": 4.71875, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/mean": 0.0003910064697265625, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/std": 0.036865234375, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/max_abs": 0.20703125, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/norm": 2.921875, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/mean": 1.2099742889404297e-05, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/std": 0.0228271484375, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/max_abs": 0.08642578125, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/norm": 3.015625, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/mean": -1.9311904907226562e-05, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/std": 0.0235595703125, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/max_abs": 0.1142578125, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_mlp_up_proj_weight/norm": 4.5, "train/tensor_param_model_layers_2_mlp_up_proj_weight/mean": 2.9325485229492188e-05, "train/tensor_param_model_layers_2_mlp_up_proj_weight/std": 0.0203857421875, "train/tensor_param_model_layers_2_mlp_up_proj_weight/max_abs": 0.0859375, "train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_mlp_down_proj_weight/norm": 4.53125, "train/tensor_param_model_layers_2_mlp_down_proj_weight/mean": -9.822845458984375e-05, "train/tensor_param_model_layers_2_mlp_down_proj_weight/std": 0.0203857421875, "train/tensor_param_model_layers_2_mlp_down_proj_weight/max_abs": 0.091796875, "train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_2_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_2_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_2_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/norm": 5.4375, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/mean": 0.0002689361572265625, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/std": 0.04248046875, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/max_abs": 0.2265625, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/norm": 4.65625, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/mean": 0.00057220458984375, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/std": 0.036376953125, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/max_abs": 0.1533203125, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/norm": 3.0625, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/mean": 4.839897155761719e-05, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/std": 0.02392578125, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/max_abs": 0.1044921875, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/norm": 3.171875, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/mean": -0.000213623046875, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/std": 0.0247802734375, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/max_abs": 0.09814453125, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_mlp_up_proj_weight/norm": 4.5, "train/tensor_param_model_layers_3_mlp_up_proj_weight/mean": -0.000148773193359375, "train/tensor_param_model_layers_3_mlp_up_proj_weight/std": 0.020263671875, "train/tensor_param_model_layers_3_mlp_up_proj_weight/max_abs": 0.083984375, "train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_mlp_down_proj_weight/norm": 4.5, "train/tensor_param_model_layers_3_mlp_down_proj_weight/mean": -0.0002155303955078125, "train/tensor_param_model_layers_3_mlp_down_proj_weight/std": 0.020263671875, "train/tensor_param_model_layers_3_mlp_down_proj_weight/max_abs": 0.0849609375, "train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_3_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_3_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_3_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/norm": 4.5625, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/mean": 0.0004425048828125, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/std": 0.03564453125, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/max_abs": 0.1494140625, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/norm": 4.46875, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/mean": 0.0002346038818359375, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/std": 0.034912109375, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/max_abs": 0.1533203125, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/norm": 2.984375, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/mean": -8.440017700195312e-05, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/std": 0.0233154296875, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/max_abs": 0.09033203125, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/norm": 3.046875, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/mean": -0.00016880035400390625, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/std": 0.0238037109375, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/max_abs": 0.10546875, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_mlp_up_proj_weight/norm": 4.46875, "train/tensor_param_model_layers_4_mlp_up_proj_weight/mean": -1.1563301086425781e-05, "train/tensor_param_model_layers_4_mlp_up_proj_weight/std": 0.020263671875, "train/tensor_param_model_layers_4_mlp_up_proj_weight/max_abs": 0.08740234375, "train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_4_mlp_down_proj_weight/mean": 6.8247318267822266e-06, "train/tensor_param_model_layers_4_mlp_down_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_4_mlp_down_proj_weight/max_abs": 0.08984375, "train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_4_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_4_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_4_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/norm": 4.84375, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/mean": -9.489059448242188e-05, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/std": 0.037841796875, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/max_abs": 0.1455078125, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/norm": 4.53125, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/mean": -0.0002918243408203125, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/std": 0.035400390625, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/max_abs": 0.1640625, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/norm": 2.921875, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/mean": -0.00018024444580078125, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/std": 0.0228271484375, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/max_abs": 0.09765625, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/norm": 3.03125, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/mean": 0.00019550323486328125, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/std": 0.023681640625, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/max_abs": 0.10888671875, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_mlp_up_proj_weight/norm": 4.5, "train/tensor_param_model_layers_5_mlp_up_proj_weight/mean": -0.0001430511474609375, "train/tensor_param_model_layers_5_mlp_up_proj_weight/std": 0.020263671875, "train/tensor_param_model_layers_5_mlp_up_proj_weight/max_abs": 0.091796875, "train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_mlp_down_proj_weight/norm": 4.40625, "train/tensor_param_model_layers_5_mlp_down_proj_weight/mean": 0.0001621246337890625, "train/tensor_param_model_layers_5_mlp_down_proj_weight/std": 0.0198974609375, "train/tensor_param_model_layers_5_mlp_down_proj_weight/max_abs": 0.0888671875, "train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_5_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_5_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_5_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/norm": 5.28125, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/mean": 6.67572021484375e-05, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/std": 0.041259765625, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/max_abs": 0.1728515625, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/norm": 5.03125, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/mean": 0.00015926361083984375, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/std": 0.039306640625, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/max_abs": 0.16796875, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/norm": 3.21875, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/mean": -0.000270843505859375, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/std": 0.025146484375, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/max_abs": 0.125, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/norm": 3.515625, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/mean": 0.0003662109375, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/std": 0.0274658203125, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/max_abs": 0.1201171875, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_mlp_up_proj_weight/norm": 4.5, "train/tensor_param_model_layers_6_mlp_up_proj_weight/mean": 8.106231689453125e-05, "train/tensor_param_model_layers_6_mlp_up_proj_weight/std": 0.020263671875, "train/tensor_param_model_layers_6_mlp_up_proj_weight/max_abs": 0.083984375, "train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_mlp_down_proj_weight/norm": 4.5, "train/tensor_param_model_layers_6_mlp_down_proj_weight/mean": 4.982948303222656e-05, "train/tensor_param_model_layers_6_mlp_down_proj_weight/std": 0.020263671875, "train/tensor_param_model_layers_6_mlp_down_proj_weight/max_abs": 0.091796875, "train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_6_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_6_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_6_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/norm": 4.84375, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/mean": -7.963180541992188e-05, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/std": 0.037841796875, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/max_abs": 0.1328125, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/norm": 4.53125, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/mean": 0.00026702880859375, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/std": 0.035400390625, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/max_abs": 0.1484375, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/norm": 3.109375, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/mean": -0.0002193450927734375, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/std": 0.0242919921875, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/max_abs": 0.125, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/norm": 3.328125, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/mean": -0.00015354156494140625, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/std": 0.0260009765625, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/max_abs": 0.11328125, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_mlp_up_proj_weight/norm": 4.59375, "train/tensor_param_model_layers_7_mlp_up_proj_weight/mean": -6.246566772460938e-05, "train/tensor_param_model_layers_7_mlp_up_proj_weight/std": 0.0206298828125, "train/tensor_param_model_layers_7_mlp_up_proj_weight/max_abs": 0.0849609375, "train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_mlp_down_proj_weight/norm": 4.625, "train/tensor_param_model_layers_7_mlp_down_proj_weight/mean": 1.9550323486328125e-05, "train/tensor_param_model_layers_7_mlp_down_proj_weight/std": 0.0208740234375, "train/tensor_param_model_layers_7_mlp_down_proj_weight/max_abs": 0.09423828125, "train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_7_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_7_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_7_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/norm": 5.65625, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/mean": 7.486343383789062e-05, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/std": 0.044189453125, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/max_abs": 0.1806640625, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/norm": 5.34375, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/mean": -0.0002288818359375, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/std": 0.041748046875, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/max_abs": 0.1748046875, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/norm": 3.265625, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/mean": -0.0002651214599609375, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/std": 0.0255126953125, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/max_abs": 0.125, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/norm": 3.953125, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/mean": -0.0001621246337890625, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/std": 0.0308837890625, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/max_abs": 0.125, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_mlp_up_proj_weight/norm": 4.875, "train/tensor_param_model_layers_8_mlp_up_proj_weight/mean": -0.000225067138671875, "train/tensor_param_model_layers_8_mlp_up_proj_weight/std": 0.02197265625, "train/tensor_param_model_layers_8_mlp_up_proj_weight/max_abs": 0.103515625, "train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_mlp_down_proj_weight/norm": 5.09375, "train/tensor_param_model_layers_8_mlp_down_proj_weight/mean": -0.00016117095947265625, "train/tensor_param_model_layers_8_mlp_down_proj_weight/std": 0.0230712890625, "train/tensor_param_model_layers_8_mlp_down_proj_weight/max_abs": 0.08642578125, "train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_8_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_8_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_8_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_norm_weight/norm": 11.3125, "train/tensor_param_model_norm_weight/mean": 1.0, "train/tensor_param_model_norm_weight/std": 0.0, "train/tensor_param_model_norm_weight/max_abs": 1.0, "train/tensor_param_model_norm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_norm_weight/frac_near_user_limit": 0.0} +{"step": 750, "epoch": 1.0107853050219076, "timestamp": 1786256369.8721795, "train_runtime": 1545.7365, "train_samples_per_second": 621.063, "train_steps_per_second": 0.485, "total_flos": 4354347480907776.0, "train_loss": 61.381723103841146, "train/total_time_seconds": 676.0385475568473, "train/time_per_step_avg": 0.9476513889431953, "train/epoch_time_elapsed": 24.92423267289996, "train/estimated_remaining_minutes": 0.0} +{"step": 750, "epoch": 1.0107853050219076, "timestamp": 1786256380.8504226, "eval_loss": 3.108103036880493, "eval_runtime": 9.5225, "eval_samples_per_second": 1000.475, "eval_steps_per_second": 12.602, "train/total_time_seconds": 676.0385475568473, "train/time_per_step_avg": 0.9476513889431953, "train/epoch_time_elapsed": 35.902476370334625, "train/estimated_remaining_minutes": 0.0} diff --git a/outio/mlp-tanh-9L_run/training_log.jsonl b/outio/mlp-tanh-9L_run/training_log.jsonl new file mode 100644 index 0000000000000000000000000000000000000000..8153c1bb8de538a20b940e2f46fdcd46cd053674 --- /dev/null +++ b/outio/mlp-tanh-9L_run/training_log.jsonl @@ -0,0 +1,2 @@ +{"step": 20, "epoch": 0.026963262554769128, "timestamp": 1786256435.8978417, "loss": 120.16063232421875, "grad_norm": 20.0, "learning_rate": 0.0005, "train/total_time_seconds": 23.21033100038767, "train/time_per_step_avg": 1.1605165500193835, "train/epoch_time_elapsed": 43.305605318397284, "train/estimated_remaining_minutes": 14.119618025235832, "train/global/act/norm": 45825.67915864439, "train/global/act/mean": 0.0009720008240165799, "train/global/act/std": 0.4056012154736404, "train/global/act/max_abs": 8.328397750854492, "train/global/act/frac_near_dtype_limit": 0.0, "train/global/act/frac_near_user_limit": 0.0, "train/global/grad/norm": 7.367828232547335, "train/global/grad/mean": 1.7360134756876538e-06, "train/global/grad/std": 0.0013023133279015384, "train/global/grad/max_abs": 0.185546875, "train/global/grad/frac_near_dtype_limit": 0.0, "train/global/grad/frac_near_user_limit": 0.0, "train/global/param/norm": 56.836357923702764, "train/global/param/mean": 0.001200859135700027, "train/global/param/std": 0.04017675470731757, "train/global/param/max_abs": 1.0, "train/global/param/frac_near_dtype_limit": 0.0, "train/global/param/frac_near_user_limit": 0.0, "train/layer__model_layers_1/param/norm": 17.933035158611606, "train/layer__model_layers_1/param/mean": 0.00150473590202153, "train/layer__model_layers_1/param/std": 0.04424885441992243, "train/layer__model_layers_1/param/max_abs": 1.0, "train/layer__model_layers_1/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_1/param/frac_near_user_limit": 0.0, "train/layer__model_layers_8/param/norm": 17.918659803138876, "train/layer__model_layers_8/param/mean": 0.001426565851696568, "train/layer__model_layers_8/param/std": 0.04421874775973169, "train/layer__model_layers_8/param/max_abs": 1.0, "train/layer__model_layers_8/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_8/param/frac_near_user_limit": 0.0, "train/layer_model_layers_8/act/norm": 14165.892651482818, "train/layer_model_layers_8/act/mean": 0.0015629919675680308, "train/layer_model_layers_8/act/std": 0.42900728525991605, "train/layer_model_layers_8/act/max_abs": 5.59375, "train/layer_model_layers_8/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_8/act/frac_near_user_limit": 0.0, "train/layer_model_layers_8/grad/norm": 1.0449912347693717, "train/layer_model_layers_8/grad/mean": -1.409528021484195e-07, "train/layer_model_layers_8/grad/std": 0.0006452044131179452, "train/layer_model_layers_8/grad/max_abs": 0.007568359375, "train/layer_model_layers_8/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_8/grad/frac_near_user_limit": 0.0, "train/layer__model_layers_7/param/norm": 17.939752417379538, "train/layer__model_layers_7/param/mean": 0.001527878497952418, "train/layer__model_layers_7/param/std": 0.044264819558247244, "train/layer__model_layers_7/param/max_abs": 1.0, "train/layer__model_layers_7/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_7/param/frac_near_user_limit": 0.0, "train/layer__model_layers_3/param/norm": 17.937513610622013, "train/layer__model_layers_3/param/mean": 0.0014466554995817996, "train/layer__model_layers_3/param/std": 0.04426201158065172, "train/layer__model_layers_3/param/max_abs": 1.0, "train/layer__model_layers_3/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_3/param/frac_near_user_limit": 0.0, "train/layer_model_layers_3/act/norm": 14054.438547142747, "train/layer_model_layers_3/act/mean": 0.002938002347946167, "train/layer_model_layers_3/act/std": 0.42579208331501933, "train/layer_model_layers_3/act/max_abs": 4.84375, "train/layer_model_layers_3/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_3/act/frac_near_user_limit": 0.0, "train/layer_model_layers_3/grad/norm": 1.7488206388186616, "train/layer_model_layers_3/grad/mean": 2.2994056518382846e-06, "train/layer_model_layers_3/grad/std": 0.0010787107165929604, "train/layer_model_layers_3/grad/max_abs": 0.01092529296875, "train/layer_model_layers_3/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_3/grad/frac_near_user_limit": 0.0, "train/layer_model_layers_7/act/norm": 14142.08189676352, "train/layer_model_layers_7/act/mean": 0.0005603524354787973, "train/layer_model_layers_7/act/std": 0.42826948381510205, "train/layer_model_layers_7/act/max_abs": 5.59375, "train/layer_model_layers_7/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_7/act/frac_near_user_limit": 0.0, "train/layer_model_layers_7/grad/norm": 1.0915601981948118, "train/layer_model_layers_7/grad/mean": -5.478754539010747e-07, "train/layer_model_layers_7/grad/std": 0.0006737248985529821, "train/layer_model_layers_7/grad/max_abs": 0.00836181640625, "train/layer_model_layers_7/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_7/grad/frac_near_user_limit": 0.0, "train/layer_model_layers_0/act/norm": 13963.784289654679, "train/layer_model_layers_0/act/mean": -0.00017583585129334376, "train/layer_model_layers_0/act/std": 0.4233368714469234, "train/layer_model_layers_0/act/max_abs": 4.75, "train/layer_model_layers_0/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_0/act/frac_near_user_limit": 0.0, "train/layer_model_layers_0/grad/norm": 4.200656564761473, "train/layer_model_layers_0/grad/mean": 5.849929514280203e-09, "train/layer_model_layers_0/grad/std": 0.0025923967714678426, "train/layer_model_layers_0/grad/max_abs": 0.037353515625, "train/layer_model_layers_0/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_0/grad/frac_near_user_limit": 0.0, "train/layer_model_layers_4/act/norm": 14065.020087062054, "train/layer_model_layers_4/act/mean": 0.001356604007574228, "train/layer_model_layers_4/act/std": 0.4259628077063781, "train/layer_model_layers_4/act/max_abs": 5.375, "train/layer_model_layers_4/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_4/act/frac_near_user_limit": 0.0, "train/layer_model_layers_4/grad/norm": 1.4648445572552842, "train/layer_model_layers_4/grad/mean": -2.1866095570524845e-06, "train/layer_model_layers_4/grad/std": 0.0009039487622426939, "train/layer_model_layers_4/grad/max_abs": 0.009765625, "train/layer_model_layers_4/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_4/grad/frac_near_user_limit": 0.0, "train/layer__model_layers_0/param/norm": 17.93527452440093, "train/layer__model_layers_0/param/mean": 0.0015731311625511895, "train/layer__model_layers_0/param/std": 0.04425206676629325, "train/layer__model_layers_0/param/max_abs": 1.0, "train/layer__model_layers_0/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_0/param/frac_near_user_limit": 0.0, "train/layer_model_layers_6/act/norm": 14112.49991284504, "train/layer_model_layers_6/act/mean": 0.0014113554587730994, "train/layer_model_layers_6/act/std": 0.42734430023274333, "train/layer_model_layers_6/act/max_abs": 5.15625, "train/layer_model_layers_6/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_6/act/frac_near_user_limit": 0.0, "train/layer_model_layers_6/grad/norm": 1.154570131496639, "train/layer_model_layers_6/grad/mean": -6.731039354637231e-07, "train/layer_model_layers_6/grad/std": 0.0007126712014498404, "train/layer_model_layers_6/grad/max_abs": 0.00775146484375, "train/layer_model_layers_6/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_6/grad/frac_near_user_limit": 0.0, "train/layer_model_layers_1/act/norm": 13992.899675692339, "train/layer_model_layers_1/act/mean": -5.987723572896077e-05, "train/layer_model_layers_1/act/std": 0.4237531971235928, "train/layer_model_layers_1/act/max_abs": 5.28125, "train/layer_model_layers_1/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_1/act/frac_near_user_limit": 0.0, "train/layer_model_layers_1/grad/norm": 2.708397429771687, "train/layer_model_layers_1/grad/mean": 2.754121628704709e-06, "train/layer_model_layers_1/grad/std": 0.0016707387393106302, "train/layer_model_layers_1/grad/max_abs": 0.0184326171875, "train/layer_model_layers_1/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_1/grad/frac_near_user_limit": 0.0, "train/layer__model_layers_6/param/norm": 17.92636985918022, "train/layer__model_layers_6/param/mean": 0.0016320834107778374, "train/layer__model_layers_6/param/std": 0.04422801786462898, "train/layer__model_layers_6/param/max_abs": 1.0, "train/layer__model_layers_6/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_6/param/frac_near_user_limit": 0.0, "train/layer__model_layers_5/param/norm": 17.927609152418373, "train/layer__model_layers_5/param/mean": 0.001554492111325078, "train/layer__model_layers_5/param/std": 0.04425301650466886, "train/layer__model_layers_5/param/max_abs": 1.0, "train/layer__model_layers_5/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_5/param/frac_near_user_limit": 0.0, "train/layer__model_layers_2/param/norm": 17.933035158611606, "train/layer__model_layers_2/param/mean": 0.0015104921671232083, "train/layer__model_layers_2/param/std": 0.044265280721252936, "train/layer__model_layers_2/param/max_abs": 1.0, "train/layer__model_layers_2/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_2/param/frac_near_user_limit": 0.0, "train/layer_model_layers_2/act/norm": 14024.082295312, "train/layer_model_layers_2/act/mean": 0.0027060508728027344, "train/layer_model_layers_2/act/std": 0.4247193027941706, "train/layer_model_layers_2/act/max_abs": 4.875, "train/layer_model_layers_2/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_2/act/frac_near_user_limit": 0.0, "train/layer_model_layers_2/grad/norm": 2.2923954362436416, "train/layer_model_layers_2/grad/mean": 2.7492822088154436e-06, "train/layer_model_layers_2/grad/std": 0.0014146170196390482, "train/layer_model_layers_2/grad/max_abs": 0.01556396484375, "train/layer_model_layers_2/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_2/grad/frac_near_user_limit": 0.0, "train/layer_model_layers_5/act/norm": 14100.803074018622, "train/layer_model_layers_5/act/mean": 0.0020468876912043644, "train/layer_model_layers_5/act/std": 0.427023307556937, "train/layer_model_layers_5/act/max_abs": 5.21875, "train/layer_model_layers_5/act/frac_near_dtype_limit": 0.0, "train/layer_model_layers_5/act/frac_near_user_limit": 0.0, "train/layer_model_layers_5/grad/norm": 1.241541045998697, "train/layer_model_layers_5/grad/mean": -1.1199119906907382e-06, "train/layer_model_layers_5/grad/std": 0.0007662412491179434, "train/layer_model_layers_5/grad/max_abs": 0.00872802734375, "train/layer_model_layers_5/grad/frac_near_dtype_limit": 0.0, "train/layer_model_layers_5/grad/frac_near_user_limit": 0.0, "train/layer__model_layers_4/param/norm": 17.9352881367118, "train/layer__model_layers_4/param/mean": 0.001547672075340045, "train/layer__model_layers_4/param/std": 0.04425290086076607, "train/layer__model_layers_4/param/max_abs": 1.0, "train/layer__model_layers_4/param/frac_near_dtype_limit": 0.0, "train/layer__model_layers_4/param/frac_near_user_limit": 0.0, "train/tensor_act_model_embed_tokens/norm": 183.14187215122737, "train/tensor_act_model_embed_tokens/mean": -3.91155481338501e-05, "train/tensor_act_model_embed_tokens/std": 0.02001953329041244, "train/tensor_act_model_embed_tokens/max_abs": 0.09619140625, "train/tensor_act_model_embed_tokens/frac_near_dtype_limit": 0.0, "train/tensor_act_model_embed_tokens/frac_near_user_limit": 0.0, "train/tensor_act_model_rotary_emb/norm": 3632.2067871093745, "train/tensor_act_model_rotary_emb/mean": 0.333984375, "train/tensor_act_model_rotary_emb/std": 0.71875, "train/tensor_act_model_rotary_emb/max_abs": 1.0, "train/tensor_act_model_rotary_emb/frac_near_dtype_limit": 0.0, "train/tensor_act_model_rotary_emb/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_input_layernorm/norm": 9146.768310589385, "train/tensor_act_model_layers_0_input_layernorm/mean": -0.0023627281188964844, "train/tensor_act_model_layers_0_input_layernorm/std": 1.000000091723332, "train/tensor_act_model_layers_0_input_layernorm/max_abs": 4.375, "train/tensor_act_model_layers_0_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_q_proj/norm": 2073.7606174205857, "train/tensor_act_model_layers_0_self_attn_q_proj/mean": -0.0006031990051269532, "train/tensor_act_model_layers_0_self_attn_q_proj/std": 0.2265625557654413, "train/tensor_act_model_layers_0_self_attn_q_proj/max_abs": 1.0703125, "train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_k_proj/norm": 2089.1339244416013, "train/tensor_act_model_layers_0_self_attn_k_proj/mean": 8.234567940235139e-05, "train/tensor_act_model_layers_0_self_attn_k_proj/std": 0.2281499629770971, "train/tensor_act_model_layers_0_self_attn_k_proj/max_abs": 1.0234375, "train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_v_proj/norm": 2107.624164923062, "train/tensor_act_model_layers_0_self_attn_v_proj/mean": -0.0013871192932128906, "train/tensor_act_model_layers_0_self_attn_v_proj/std": 0.23046878774569188, "train/tensor_act_model_layers_0_self_attn_v_proj/max_abs": 1.0703125, "train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_o_proj/norm": 78.02399889508851, "train/tensor_act_model_layers_0_self_attn_o_proj/mean": 0.0001543760299682617, "train/tensor_act_model_layers_0_self_attn_o_proj/std": 0.008522998259386573, "train/tensor_act_model_layers_0_self_attn_o_proj/max_abs": 0.2578125, "train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_self_attn/norm": 78.02399889508851, "train/tensor_act_model_layers_0_self_attn/mean": 0.0001543760299682617, "train/tensor_act_model_layers_0_self_attn/std": 0.008522998259386573, "train/tensor_act_model_layers_0_self_attn/max_abs": 0.2578125, "train/tensor_act_model_layers_0_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_post_attention_layernorm/norm": 9148.850097786815, "train/tensor_act_model_layers_0_post_attention_layernorm/mean": 0.004792213439941406, "train/tensor_act_model_layers_0_post_attention_layernorm/std": 1.000002007208682, "train/tensor_act_model_layers_0_post_attention_layernorm/max_abs": 4.75, "train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_mlp_up_proj/norm": 3560.757550565122, "train/tensor_act_model_layers_0_mlp_up_proj/mean": -7.581710815429688e-05, "train/tensor_act_model_layers_0_mlp_up_proj/std": 0.2246094857487788, "train/tensor_act_model_layers_0_mlp_up_proj/max_abs": 1.1796875, "train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_mlp_down_proj/norm": 771.7085156076147, "train/tensor_act_model_layers_0_mlp_down_proj/mean": -0.001001596450805664, "train/tensor_act_model_layers_0_mlp_down_proj/std": 0.08425967874819335, "train/tensor_act_model_layers_0_mlp_down_proj/max_abs": 0.490234375, "train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0_mlp/norm": 771.7085156076147, "train/tensor_act_model_layers_0_mlp/mean": -0.001001596450805664, "train/tensor_act_model_layers_0_mlp/std": 0.08425967874819335, "train/tensor_act_model_layers_0_mlp/max_abs": 0.490234375, "train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_0/norm": 795.7147017678604, "train/tensor_act_model_layers_0/mean": -0.0008854866027832031, "train/tensor_act_model_layers_0/std": 0.0869143314307497, "train/tensor_act_model_layers_0/max_abs": 0.50390625, "train/tensor_act_model_layers_0/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_0/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_input_layernorm/norm": 9158.302429209018, "train/tensor_act_model_layers_1_input_layernorm/mean": -0.0088043212890625, "train/tensor_act_model_layers_1_input_layernorm/std": 1.000003035874154, "train/tensor_act_model_layers_1_input_layernorm/max_abs": 5.28125, "train/tensor_act_model_layers_1_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_q_proj/norm": 2035.2156649415904, "train/tensor_act_model_layers_1_self_attn_q_proj/mean": -0.00579071044921875, "train/tensor_act_model_layers_1_self_attn_q_proj/std": 0.22210801187994125, "train/tensor_act_model_layers_1_self_attn_q_proj/max_abs": 1.1875, "train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_k_proj/norm": 2082.805685076487, "train/tensor_act_model_layers_1_self_attn_k_proj/mean": 0.0003761379048228264, "train/tensor_act_model_layers_1_self_attn_k_proj/std": 0.22747849920145083, "train/tensor_act_model_layers_1_self_attn_k_proj/max_abs": 1.1640625, "train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_v_proj/norm": 2049.253818858697, "train/tensor_act_model_layers_1_self_attn_v_proj/mean": -0.0009049177169799805, "train/tensor_act_model_layers_1_self_attn_v_proj/std": 0.22363317047832146, "train/tensor_act_model_layers_1_self_attn_v_proj/max_abs": 1.1796875, "train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_o_proj/norm": 180.36924706885924, "train/tensor_act_model_layers_1_self_attn_o_proj/mean": 0.0011792182922363281, "train/tensor_act_model_layers_1_self_attn_o_proj/std": 0.01964113700617179, "train/tensor_act_model_layers_1_self_attn_o_proj/max_abs": 0.2412109375, "train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_self_attn/norm": 180.36924706885924, "train/tensor_act_model_layers_1_self_attn/mean": 0.0011792182922363281, "train/tensor_act_model_layers_1_self_attn/std": 0.01964113700617179, "train/tensor_act_model_layers_1_self_attn/max_abs": 0.2412109375, "train/tensor_act_model_layers_1_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_post_attention_layernorm/norm": 9158.334106452012, "train/tensor_act_model_layers_1_post_attention_layernorm/mean": 0.004477977752685548, "train/tensor_act_model_layers_1_post_attention_layernorm/std": 1.0000032041136646, "train/tensor_act_model_layers_1_post_attention_layernorm/max_abs": 5.28125, "train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_mlp_up_proj/norm": 3579.792003483908, "train/tensor_act_model_layers_1_mlp_up_proj/mean": 0.001192331314086914, "train/tensor_act_model_layers_1_mlp_up_proj/std": 0.22558610679893412, "train/tensor_act_model_layers_1_mlp_up_proj/max_abs": 1.1796875, "train/tensor_act_model_layers_1_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_mlp_down_proj/norm": 783.6076718471597, "train/tensor_act_model_layers_1_mlp_down_proj/mean": 0.0012128353118896484, "train/tensor_act_model_layers_1_mlp_down_proj/std": 0.08551056605861647, "train/tensor_act_model_layers_1_mlp_down_proj/max_abs": 0.453125, "train/tensor_act_model_layers_1_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1_mlp/norm": 783.6076718471597, "train/tensor_act_model_layers_1_mlp/mean": 0.0012128353118896484, "train/tensor_act_model_layers_1_mlp/std": 0.08551056605861647, "train/tensor_act_model_layers_1_mlp/max_abs": 0.453125, "train/tensor_act_model_layers_1_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_1/norm": 1124.2650574824988, "train/tensor_act_model_layers_1/mean": 0.0015063285827636719, "train/tensor_act_model_layers_1/std": 0.12271151907771602, "train/tensor_act_model_layers_1/max_abs": 0.609375, "train/tensor_act_model_layers_1/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_1/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_input_layernorm/norm": 9158.614257822699, "train/tensor_act_model_layers_2_input_layernorm/mean": 0.01171112060546875, "train/tensor_act_model_layers_2_input_layernorm/std": 1.0000010819343121, "train/tensor_act_model_layers_2_input_layernorm/max_abs": 4.875, "train/tensor_act_model_layers_2_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_q_proj/norm": 2065.720629302085, "train/tensor_act_model_layers_2_self_attn_q_proj/mean": -0.012233734130859375, "train/tensor_act_model_layers_2_self_attn_q_proj/std": 0.22515963858329976, "train/tensor_act_model_layers_2_self_attn_q_proj/max_abs": 1.3828125, "train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_k_proj/norm": 2092.7739870028217, "train/tensor_act_model_layers_2_self_attn_k_proj/mean": 0.008234024047851562, "train/tensor_act_model_layers_2_self_attn_k_proj/std": 0.22839403667703648, "train/tensor_act_model_layers_2_self_attn_k_proj/max_abs": 1.3046875, "train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_v_proj/norm": 2097.41569083777, "train/tensor_act_model_layers_2_self_attn_v_proj/mean": -0.008272171020507812, "train/tensor_act_model_layers_2_self_attn_v_proj/std": 0.228882807276572, "train/tensor_act_model_layers_2_self_attn_v_proj/max_abs": 1.234375, "train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_o_proj/norm": 228.47226970029095, "train/tensor_act_model_layers_2_self_attn_o_proj/mean": 0.0012559890747070312, "train/tensor_act_model_layers_2_self_attn_o_proj/std": 0.024912951683477253, "train/tensor_act_model_layers_2_self_attn_o_proj/max_abs": 0.2265625, "train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_self_attn/norm": 228.47226970029095, "train/tensor_act_model_layers_2_self_attn/mean": 0.0012559890747070312, "train/tensor_act_model_layers_2_self_attn/std": 0.024912951683477253, "train/tensor_act_model_layers_2_self_attn/max_abs": 0.2265625, "train/tensor_act_model_layers_2_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_post_attention_layernorm/norm": 9158.621521007424, "train/tensor_act_model_layers_2_post_attention_layernorm/mean": 0.021453857421875, "train/tensor_act_model_layers_2_post_attention_layernorm/std": 1.0000019553098334, "train/tensor_act_model_layers_2_post_attention_layernorm/max_abs": 4.875, "train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_mlp_up_proj/norm": 3570.1849654717043, "train/tensor_act_model_layers_2_mlp_up_proj/mean": 0.0017571449279785156, "train/tensor_act_model_layers_2_mlp_up_proj/std": 0.2249152545309565, "train/tensor_act_model_layers_2_mlp_up_proj/max_abs": 1.2734375, "train/tensor_act_model_layers_2_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_mlp_down_proj/norm": 753.8613409709429, "train/tensor_act_model_layers_2_mlp_down_proj/mean": 0.0012459754943847656, "train/tensor_act_model_layers_2_mlp_down_proj/std": 0.082367886029098, "train/tensor_act_model_layers_2_mlp_down_proj/max_abs": 0.443359375, "train/tensor_act_model_layers_2_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2_mlp/norm": 753.8613409709429, "train/tensor_act_model_layers_2_mlp/mean": 0.0012459754943847656, "train/tensor_act_model_layers_2_mlp/std": 0.082367886029098, "train/tensor_act_model_layers_2_mlp/max_abs": 0.443359375, "train/tensor_act_model_layers_2_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_2/norm": 1371.506467834684, "train/tensor_act_model_layers_2/mean": 0.004010200500488282, "train/tensor_act_model_layers_2/std": 0.14959821173642976, "train/tensor_act_model_layers_2/max_abs": 0.79296875, "train/tensor_act_model_layers_2/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_2/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_input_layernorm/norm": 9158.712524420385, "train/tensor_act_model_layers_3_input_layernorm/mean": 0.02681732177734375, "train/tensor_act_model_layers_3_input_layernorm/std": 1.0000036635543632, "train/tensor_act_model_layers_3_input_layernorm/max_abs": 4.75, "train/tensor_act_model_layers_3_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_q_proj/norm": 2092.7443792291797, "train/tensor_act_model_layers_3_self_attn_q_proj/mean": 0.0022826194763183594, "train/tensor_act_model_layers_3_self_attn_q_proj/std": 0.22851674734790905, "train/tensor_act_model_layers_3_self_attn_q_proj/max_abs": 1.1875, "train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_k_proj/norm": 2053.7752410220805, "train/tensor_act_model_layers_3_self_attn_k_proj/mean": -0.00013938546180725098, "train/tensor_act_model_layers_3_self_attn_k_proj/std": 0.22442723998789932, "train/tensor_act_model_layers_3_self_attn_k_proj/max_abs": 1.1484375, "train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_v_proj/norm": 2072.680207254902, "train/tensor_act_model_layers_3_self_attn_v_proj/mean": -0.006090164184570312, "train/tensor_act_model_layers_3_self_attn_v_proj/std": 0.22631944523624664, "train/tensor_act_model_layers_3_self_attn_v_proj/max_abs": 1.1796875, "train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_o_proj/norm": 226.46827816100497, "train/tensor_act_model_layers_3_self_attn_o_proj/mean": 0.0013756752014160156, "train/tensor_act_model_layers_3_self_attn_o_proj/std": 0.02468457451209307, "train/tensor_act_model_layers_3_self_attn_o_proj/max_abs": 0.2265625, "train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_self_attn/norm": 226.46827816100497, "train/tensor_act_model_layers_3_self_attn/mean": 0.0013756752014160156, "train/tensor_act_model_layers_3_self_attn/std": 0.02468457451209307, "train/tensor_act_model_layers_3_self_attn/max_abs": 0.2265625, "train/tensor_act_model_layers_3_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_post_attention_layernorm/norm": 9158.71673584305, "train/tensor_act_model_layers_3_post_attention_layernorm/mean": 0.03549957275390625, "train/tensor_act_model_layers_3_post_attention_layernorm/std": 1.000005643071284, "train/tensor_act_model_layers_3_post_attention_layernorm/max_abs": 4.84375, "train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_mlp_up_proj/norm": 3595.762705617342, "train/tensor_act_model_layers_3_mlp_up_proj/mean": -0.0061511993408203125, "train/tensor_act_model_layers_3_mlp_up_proj/std": 0.22656262048581174, "train/tensor_act_model_layers_3_mlp_up_proj/max_abs": 1.28125, "train/tensor_act_model_layers_3_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_mlp_down_proj/norm": 791.0893592431163, "train/tensor_act_model_layers_3_mlp_down_proj/mean": -0.0032863616943359375, "train/tensor_act_model_layers_3_mlp_down_proj/std": 0.08636519797159739, "train/tensor_act_model_layers_3_mlp_down_proj/max_abs": 0.416015625, "train/tensor_act_model_layers_3_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3_mlp/norm": 791.0893592431163, "train/tensor_act_model_layers_3_mlp/mean": -0.0032863616943359375, "train/tensor_act_model_layers_3_mlp/std": 0.08636519797159739, "train/tensor_act_model_layers_3_mlp/max_abs": 0.416015625, "train/tensor_act_model_layers_3_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_3/norm": 1608.0517896816252, "train/tensor_act_model_layers_3/mean": 0.002099037170410156, "train/tensor_act_model_layers_3/std": 0.17572100292915102, "train/tensor_act_model_layers_3/max_abs": 0.95703125, "train/tensor_act_model_layers_3/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_3/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_input_layernorm/norm": 9158.770935065297, "train/tensor_act_model_layers_4_input_layernorm/mean": 0.011629104614257812, "train/tensor_act_model_layers_4_input_layernorm/std": 1.0000035008045074, "train/tensor_act_model_layers_4_input_layernorm/max_abs": 5.1875, "train/tensor_act_model_layers_4_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_q_proj/norm": 2065.6788890504995, "train/tensor_act_model_layers_4_self_attn_q_proj/mean": -0.009613037109375, "train/tensor_act_model_layers_4_self_attn_q_proj/std": 0.22534347173992078, "train/tensor_act_model_layers_4_self_attn_q_proj/max_abs": 1.1875, "train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_k_proj/norm": 2100.7299753273983, "train/tensor_act_model_layers_4_self_attn_k_proj/mean": 0.006021499633789062, "train/tensor_act_model_layers_4_self_attn_k_proj/std": 0.2294319808690155, "train/tensor_act_model_layers_4_self_attn_k_proj/max_abs": 1.1953125, "train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_v_proj/norm": 2030.936626263437, "train/tensor_act_model_layers_4_self_attn_v_proj/mean": -0.00487518310546875, "train/tensor_act_model_layers_4_self_attn_v_proj/std": 0.22168016473050778, "train/tensor_act_model_layers_4_self_attn_v_proj/max_abs": 1.1640625, "train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_o_proj/norm": 255.11202588150823, "train/tensor_act_model_layers_4_self_attn_o_proj/mean": 0.0005696415901184081, "train/tensor_act_model_layers_4_self_attn_o_proj/std": 0.027843332063912866, "train/tensor_act_model_layers_4_self_attn_o_proj/max_abs": 0.236328125, "train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_self_attn/norm": 255.11202588150823, "train/tensor_act_model_layers_4_self_attn/mean": 0.0005696415901184081, "train/tensor_act_model_layers_4_self_attn/std": 0.027843332063912866, "train/tensor_act_model_layers_4_self_attn/max_abs": 0.236328125, "train/tensor_act_model_layers_4_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_post_attention_layernorm/norm": 9158.769226082162, "train/tensor_act_model_layers_4_post_attention_layernorm/mean": 0.01483917236328125, "train/tensor_act_model_layers_4_post_attention_layernorm/std": 1.0000021569116524, "train/tensor_act_model_layers_4_post_attention_layernorm/max_abs": 5.375, "train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_mlp_up_proj/norm": 3555.220216731526, "train/tensor_act_model_layers_4_mlp_up_proj/mean": -6.467103958129883e-06, "train/tensor_act_model_layers_4_mlp_up_proj/std": 0.22412190403975368, "train/tensor_act_model_layers_4_mlp_up_proj/max_abs": 1.234375, "train/tensor_act_model_layers_4_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_mlp_down_proj/norm": 773.9119460845479, "train/tensor_act_model_layers_4_mlp_down_proj/mean": -0.0013844966888427734, "train/tensor_act_model_layers_4_mlp_down_proj/std": 0.0845341598013693, "train/tensor_act_model_layers_4_mlp_down_proj/max_abs": 0.4609375, "train/tensor_act_model_layers_4_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4_mlp/norm": 773.9119460845479, "train/tensor_act_model_layers_4_mlp/mean": -0.0013844966888427734, "train/tensor_act_model_layers_4_mlp/std": 0.0845341598013693, "train/tensor_act_model_layers_4_mlp/max_abs": 0.4609375, "train/tensor_act_model_layers_4_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_4/norm": 1812.7927373217005, "train/tensor_act_model_layers_4/mean": 0.001283407211303711, "train/tensor_act_model_layers_4/std": 0.19793796607985248, "train/tensor_act_model_layers_4/max_abs": 1.109375, "train/tensor_act_model_layers_4/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_4/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_input_layernorm/norm": 9158.807006839641, "train/tensor_act_model_layers_5_input_layernorm/mean": 0.006045341491699219, "train/tensor_act_model_layers_5_input_layernorm/std": 1.000002211309776, "train/tensor_act_model_layers_5_input_layernorm/max_abs": 5.1875, "train/tensor_act_model_layers_5_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_q_proj/norm": 2103.186552565286, "train/tensor_act_model_layers_5_self_attn_q_proj/mean": 0.011692047119140625, "train/tensor_act_model_layers_5_self_attn_q_proj/std": 0.22937136367933203, "train/tensor_act_model_layers_5_self_attn_q_proj/max_abs": 1.203125, "train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_k_proj/norm": 2051.8907096354474, "train/tensor_act_model_layers_5_self_attn_k_proj/mean": 0.0010291337966918945, "train/tensor_act_model_layers_5_self_attn_k_proj/std": 0.22399984632619202, "train/tensor_act_model_layers_5_self_attn_k_proj/max_abs": 1.171875, "train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_v_proj/norm": 2078.0090258486925, "train/tensor_act_model_layers_5_self_attn_v_proj/mean": 9.191036224365232e-05, "train/tensor_act_model_layers_5_self_attn_v_proj/std": 0.22674686727818974, "train/tensor_act_model_layers_5_self_attn_v_proj/max_abs": 1.2578125, "train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_o_proj/norm": 259.09118251663875, "train/tensor_act_model_layers_5_self_attn_o_proj/mean": 0.0007239580154418945, "train/tensor_act_model_layers_5_self_attn_o_proj/std": 0.028286906480645278, "train/tensor_act_model_layers_5_self_attn_o_proj/max_abs": 0.2197265625, "train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_self_attn/norm": 259.09118251663875, "train/tensor_act_model_layers_5_self_attn/mean": 0.0007239580154418945, "train/tensor_act_model_layers_5_self_attn/std": 0.028286906480645278, "train/tensor_act_model_layers_5_self_attn/max_abs": 0.2197265625, "train/tensor_act_model_layers_5_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_post_attention_layernorm/norm": 9158.804016119773, "train/tensor_act_model_layers_5_post_attention_layernorm/mean": 0.00972747802734375, "train/tensor_act_model_layers_5_post_attention_layernorm/std": 1.000002252895874, "train/tensor_act_model_layers_5_post_attention_layernorm/max_abs": 5.21875, "train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_mlp_up_proj/norm": 3573.309552268948, "train/tensor_act_model_layers_5_mlp_up_proj/mean": -0.001348257064819336, "train/tensor_act_model_layers_5_mlp_up_proj/std": 0.2254035162165185, "train/tensor_act_model_layers_5_mlp_up_proj/max_abs": 1.171875, "train/tensor_act_model_layers_5_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_mlp_down_proj/norm": 761.6220634845612, "train/tensor_act_model_layers_5_mlp_down_proj/mean": -0.00046265125274658203, "train/tensor_act_model_layers_5_mlp_down_proj/std": 0.08306951397613167, "train/tensor_act_model_layers_5_mlp_down_proj/max_abs": 0.439453125, "train/tensor_act_model_layers_5_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5_mlp/norm": 761.6220634845612, "train/tensor_act_model_layers_5_mlp/mean": -0.00046265125274658203, "train/tensor_act_model_layers_5_mlp/std": 0.08306951397613167, "train/tensor_act_model_layers_5_mlp/max_abs": 0.439453125, "train/tensor_act_model_layers_5_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_5/norm": 2012.5868662278626, "train/tensor_act_model_layers_5/mean": 0.0015457868576049805, "train/tensor_act_model_layers_5/std": 0.21972739454715445, "train/tensor_act_model_layers_5/max_abs": 1.2109375, "train/tensor_act_model_layers_5/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_5/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_input_layernorm/norm": 9158.826354986366, "train/tensor_act_model_layers_6_input_layernorm/mean": 0.007315635681152344, "train/tensor_act_model_layers_6_input_layernorm/std": 1.0000025588923567, "train/tensor_act_model_layers_6_input_layernorm/max_abs": 5.15625, "train/tensor_act_model_layers_6_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_q_proj/norm": 2023.8789405812654, "train/tensor_act_model_layers_6_self_attn_q_proj/mean": 0.008485794067382814, "train/tensor_act_model_layers_6_self_attn_q_proj/std": 0.22082680571111493, "train/tensor_act_model_layers_6_self_attn_q_proj/max_abs": 1.203125, "train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_k_proj/norm": 2052.806148941093, "train/tensor_act_model_layers_6_self_attn_k_proj/mean": -0.011730194091796875, "train/tensor_act_model_layers_6_self_attn_k_proj/std": 0.22381882679304918, "train/tensor_act_model_layers_6_self_attn_k_proj/max_abs": 1.1328125, "train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_v_proj/norm": 2090.723313709615, "train/tensor_act_model_layers_6_self_attn_v_proj/mean": 0.014347076416015625, "train/tensor_act_model_layers_6_self_attn_v_proj/std": 0.22772402108894638, "train/tensor_act_model_layers_6_self_attn_v_proj/max_abs": 1.28125, "train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_o_proj/norm": 253.97356640288828, "train/tensor_act_model_layers_6_self_attn_o_proj/mean": 0.0006264448165893555, "train/tensor_act_model_layers_6_self_attn_o_proj/std": 0.027715097177038607, "train/tensor_act_model_layers_6_self_attn_o_proj/max_abs": 0.2177734375, "train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_self_attn/norm": 253.97356640288828, "train/tensor_act_model_layers_6_self_attn/mean": 0.0006264448165893555, "train/tensor_act_model_layers_6_self_attn/std": 0.027715097177038607, "train/tensor_act_model_layers_6_self_attn/max_abs": 0.2177734375, "train/tensor_act_model_layers_6_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_post_attention_layernorm/norm": 9158.82855225413, "train/tensor_act_model_layers_6_post_attention_layernorm/mean": 0.010058403015136719, "train/tensor_act_model_layers_6_post_attention_layernorm/std": 1.0000044143605804, "train/tensor_act_model_layers_6_post_attention_layernorm/max_abs": 5.09375, "train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_mlp_up_proj/norm": 3566.9119113736056, "train/tensor_act_model_layers_6_mlp_up_proj/mean": -0.004803657531738281, "train/tensor_act_model_layers_6_mlp_up_proj/std": 0.22460983234978, "train/tensor_act_model_layers_6_mlp_up_proj/max_abs": 1.21875, "train/tensor_act_model_layers_6_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_mlp_down_proj/norm": 783.6505542822516, "train/tensor_act_model_layers_6_mlp_down_proj/mean": 0.0002854466438293457, "train/tensor_act_model_layers_6_mlp_down_proj/std": 0.08557226024157712, "train/tensor_act_model_layers_6_mlp_down_proj/max_abs": 0.5078125, "train/tensor_act_model_layers_6_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6_mlp/norm": 783.6505542822516, "train/tensor_act_model_layers_6_mlp/mean": 0.0002854466438293457, "train/tensor_act_model_layers_6_mlp/std": 0.08557226024157712, "train/tensor_act_model_layers_6_mlp/max_abs": 0.5078125, "train/tensor_act_model_layers_6_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_6/norm": 2152.4655637454157, "train/tensor_act_model_layers_6/mean": 0.0024580955505371094, "train/tensor_act_model_layers_6/std": 0.2349867621171829, "train/tensor_act_model_layers_6/max_abs": 1.46875, "train/tensor_act_model_layers_6/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_6/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_input_layernorm/norm": 9158.8432006901, "train/tensor_act_model_layers_7_input_layernorm/mean": 0.0102386474609375, "train/tensor_act_model_layers_7_input_layernorm/std": 1.0000039367509814, "train/tensor_act_model_layers_7_input_layernorm/max_abs": 5.59375, "train/tensor_act_model_layers_7_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_q_proj/norm": 2094.071764344879, "train/tensor_act_model_layers_7_self_attn_q_proj/mean": -0.0017673373222351074, "train/tensor_act_model_layers_7_self_attn_q_proj/std": 0.22870093704120772, "train/tensor_act_model_layers_7_self_attn_q_proj/max_abs": 1.125, "train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_k_proj/norm": 2070.8650656116906, "train/tensor_act_model_layers_7_self_attn_k_proj/mean": -0.0006618350744247435, "train/tensor_act_model_layers_7_self_attn_k_proj/std": 0.22613651179925, "train/tensor_act_model_layers_7_self_attn_k_proj/max_abs": 1.2265625, "train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_v_proj/norm": 2033.2786938621066, "train/tensor_act_model_layers_7_self_attn_v_proj/mean": -0.0056133270263671875, "train/tensor_act_model_layers_7_self_attn_v_proj/std": 0.22186457135269577, "train/tensor_act_model_layers_7_self_attn_v_proj/max_abs": 1.1953125, "train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_o_proj/norm": 249.8859830154156, "train/tensor_act_model_layers_7_self_attn_o_proj/mean": -0.0016374588012695312, "train/tensor_act_model_layers_7_self_attn_o_proj/std": 0.02723244874144052, "train/tensor_act_model_layers_7_self_attn_o_proj/max_abs": 0.23046875, "train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_self_attn/norm": 249.8859830154156, "train/tensor_act_model_layers_7_self_attn/mean": -0.0016374588012695312, "train/tensor_act_model_layers_7_self_attn/std": 0.02723244874144052, "train/tensor_act_model_layers_7_self_attn/max_abs": 0.23046875, "train/tensor_act_model_layers_7_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_post_attention_layernorm/norm": 9158.836975108354, "train/tensor_act_model_layers_7_post_attention_layernorm/mean": 0.0032741427421569824, "train/tensor_act_model_layers_7_post_attention_layernorm/std": 1.0000039266957168, "train/tensor_act_model_layers_7_post_attention_layernorm/max_abs": 5.4375, "train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_mlp_up_proj/norm": 3593.418878817968, "train/tensor_act_model_layers_7_mlp_up_proj/mean": -0.0003441423177719116, "train/tensor_act_model_layers_7_mlp_up_proj/std": 0.22656270507346202, "train/tensor_act_model_layers_7_mlp_up_proj/max_abs": 1.2578125, "train/tensor_act_model_layers_7_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_mlp_down_proj/norm": 753.447170392316, "train/tensor_act_model_layers_7_mlp_down_proj/mean": 0.0017676353454589844, "train/tensor_act_model_layers_7_mlp_down_proj/std": 0.08227610630735949, "train/tensor_act_model_layers_7_mlp_down_proj/max_abs": 0.4453125, "train/tensor_act_model_layers_7_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7_mlp/norm": 753.447170392316, "train/tensor_act_model_layers_7_mlp/mean": 0.0017676353454589844, "train/tensor_act_model_layers_7_mlp/std": 0.08227610630735949, "train/tensor_act_model_layers_7_mlp/max_abs": 0.4453125, "train/tensor_act_model_layers_7_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_7/norm": 2291.0751662508405, "train/tensor_act_model_layers_7/mean": 0.0025863647460937496, "train/tensor_act_model_layers_7/std": 0.2500623991634354, "train/tensor_act_model_layers_7/max_abs": 1.53125, "train/tensor_act_model_layers_7/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_7/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_input_layernorm/norm": 9158.843505887427, "train/tensor_act_model_layers_8_input_layernorm/mean": 0.010530471801757814, "train/tensor_act_model_layers_8_input_layernorm/std": 1.0000030592426397, "train/tensor_act_model_layers_8_input_layernorm/max_abs": 5.59375, "train/tensor_act_model_layers_8_input_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_input_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_q_proj/norm": 2054.917372023513, "train/tensor_act_model_layers_8_self_attn_q_proj/mean": -0.0016758441925048828, "train/tensor_act_model_layers_8_self_attn_q_proj/std": 0.22436635536725583, "train/tensor_act_model_layers_8_self_attn_q_proj/max_abs": 1.1328125, "train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_k_proj/norm": 2066.244783511504, "train/tensor_act_model_layers_8_self_attn_k_proj/mean": -0.0036034584045410156, "train/tensor_act_model_layers_8_self_attn_k_proj/std": 0.22552669692388125, "train/tensor_act_model_layers_8_self_attn_k_proj/max_abs": 1.3046875, "train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_v_proj/norm": 2069.6288876369786, "train/tensor_act_model_layers_8_self_attn_v_proj/mean": 0.008531570434570312, "train/tensor_act_model_layers_8_self_attn_v_proj/std": 0.22577102326355847, "train/tensor_act_model_layers_8_self_attn_v_proj/max_abs": 1.2109375, "train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_o_proj/norm": 258.71556009674106, "train/tensor_act_model_layers_8_self_attn_o_proj/mean": 0.0008984804153442383, "train/tensor_act_model_layers_8_self_attn_o_proj/std": 0.028224869081951577, "train/tensor_act_model_layers_8_self_attn_o_proj/max_abs": 0.25, "train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_self_attn/norm": 258.71556009674106, "train/tensor_act_model_layers_8_self_attn/mean": 0.0008984804153442383, "train/tensor_act_model_layers_8_self_attn/std": 0.028224869081951577, "train/tensor_act_model_layers_8_self_attn/max_abs": 0.25, "train/tensor_act_model_layers_8_self_attn/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_self_attn/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_post_attention_layernorm/norm": 9158.859252948552, "train/tensor_act_model_layers_8_post_attention_layernorm/mean": 0.01407623291015625, "train/tensor_act_model_layers_8_post_attention_layernorm/std": 1.000003191608266, "train/tensor_act_model_layers_8_post_attention_layernorm/max_abs": 5.53125, "train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_mlp_up_proj/norm": 3582.6683128871264, "train/tensor_act_model_layers_8_mlp_up_proj/mean": 0.000231027603149414, "train/tensor_act_model_layers_8_mlp_up_proj/std": 0.22589204684967681, "train/tensor_act_model_layers_8_mlp_up_proj/max_abs": 1.3828125, "train/tensor_act_model_layers_8_mlp_up_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_mlp_up_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_mlp_down_proj/norm": 772.4732071678798, "train/tensor_act_model_layers_8_mlp_down_proj/mean": -0.004505157470703125, "train/tensor_act_model_layers_8_mlp_down_proj/std": 0.08425981926404569, "train/tensor_act_model_layers_8_mlp_down_proj/max_abs": 0.5, "train/tensor_act_model_layers_8_mlp_down_proj/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_mlp_down_proj/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8_mlp/norm": 772.4732071678798, "train/tensor_act_model_layers_8_mlp/mean": -0.004505157470703125, "train/tensor_act_model_layers_8_mlp/std": 0.08425981926404569, "train/tensor_act_model_layers_8_mlp/max_abs": 0.5, "train/tensor_act_model_layers_8_mlp/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8_mlp/frac_near_user_limit": 0.0, "train/tensor_act_model_layers_8/norm": 2442.3700282883, "train/tensor_act_model_layers_8/mean": -0.001019805669784546, "train/tensor_act_model_layers_8/std": 0.2667264865294788, "train/tensor_act_model_layers_8/max_abs": 1.5546875, "train/tensor_act_model_layers_8/frac_near_dtype_limit": 0.0, "train/tensor_act_model_layers_8/frac_near_user_limit": 0.0, "train/tensor_act_model_norm/norm": 9158.858032251153, "train/tensor_act_model_norm/mean": -0.0038902759552001953, "train/tensor_act_model_norm/std": 1.000004099419111, "train/tensor_act_model_norm/max_abs": 5.34375, "train/tensor_act_model_norm/frac_near_dtype_limit": 0.0, "train/tensor_act_model_norm/frac_near_user_limit": 0.0, "train/tensor_act_model/norm": 9158.858032251153, "train/tensor_act_model/mean": -0.0038902759552001953, "train/tensor_act_model/std": 1.000004099419111, "train/tensor_act_model/max_abs": 5.34375, "train/tensor_act_model/frac_near_dtype_limit": 0.0, "train/tensor_act_model/frac_near_user_limit": 0.0, "train/tensor_act_lm_head/norm": 11726.645165878956, "train/tensor_act_lm_head/mean": -0.0027561187744140625, "train/tensor_act_lm_head/std": 0.22656261470519184, "train/tensor_act_lm_head/max_abs": 1.3203125, "train/tensor_act_lm_head/frac_near_dtype_limit": 0.0, "train/tensor_act_lm_head/frac_near_user_limit": 0.0, "train/tensor_act_/norm": 33.30498616499406, "train/tensor_act_/mean": 8.326246440410616, "train/tensor_act_/std": 0.0, "train/tensor_act_/max_abs": 8.328397750854492, "train/tensor_act_/frac_near_dtype_limit": 0.0, "train/tensor_act_/frac_near_user_limit": 0.0, "train/tensor_grad_model_norm_weight/norm": 0.04232738579843535, "train/tensor_grad_model_norm_weight/mean": 0.00026679039001464844, "train/tensor_grad_model_norm_weight/std": 0.0009002336688778526, "train/tensor_grad_model_norm_weight/max_abs": 0.0029296875, "train/tensor_grad_model_norm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_norm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/norm": 0.6664715322380179, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/mean": -8.720671758055687e-07, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/std": 0.000751989631097554, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/max_abs": 0.006103515625, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/norm": 0.6598499196816698, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/mean": 8.533243089914326e-08, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/std": 0.0007443170822539778, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/max_abs": 0.007568359375, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/norm": 0.013838451457588039, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/mean": -2.2135674953460693e-05, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/std": 0.00030603420712215727, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/max_abs": 0.00194549560546875, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/norm": 0.3149092310357127, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/mean": 1.914799213409424e-06, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/std": 0.0006155200305108026, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/max_abs": 0.00531005859375, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/norm": 0.3361567292025176, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/mean": -7.83504219725728e-07, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/std": 0.0006565563366839895, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/max_abs": 0.00640869140625, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/norm": 0.0017955851347056107, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/mean": -3.0529918149113655e-08, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/std": 3.507007765161376e-06, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/max_abs": 4.172325134277344e-05, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/norm": 0.0018480180721657947, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/mean": 1.8830178305506706e-08, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/std": 3.609412057366975e-06, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/max_abs": 3.528594970703125e-05, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_8_input_layernorm_weight/norm": 0.007290304662027054, "train/tensor_grad_model_layers_8_input_layernorm_weight/mean": 2.3213215172290802e-07, "train/tensor_grad_model_layers_8_input_layernorm_weight/std": 0.000161797306189962, "train/tensor_grad_model_layers_8_input_layernorm_weight/max_abs": 0.00074005126953125, "train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/norm": 0.7253473436367561, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/mean": -3.7022982724010944e-07, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/std": 0.0008175833184701628, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/max_abs": 0.00604248046875, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/norm": 0.6655192832990424, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/mean": -4.2983447201550007e-07, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/std": 0.0007510421049635982, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/max_abs": 0.00836181640625, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/norm": 0.013077298558122497, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/mean": 1.7954735085368156e-06, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/std": 0.00029015083781547014, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/max_abs": 0.001220703125, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/norm": 0.32286321546290747, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/mean": -1.507229171693325e-06, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/std": 0.0006305925327723769, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/max_abs": 0.0045166015625, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/norm": 0.34347939677410166, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/mean": -1.7476268112659454e-06, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/std": 0.0006708584204051948, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/max_abs": 0.00537109375, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/norm": 0.00228196070448423, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/mean": -4.3190084397792816e-08, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/std": 4.456957275638027e-06, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/max_abs": 5.6743621826171875e-05, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/norm": 0.0026383203040014954, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/mean": 2.7741862140828744e-09, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/std": 5.15297703171949e-06, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/max_abs": 6.723403930664062e-05, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_7_input_layernorm_weight/norm": 0.007552827001683844, "train/tensor_grad_model_layers_7_input_layernorm_weight/mean": 2.4847686290740967e-05, "train/tensor_grad_model_layers_7_input_layernorm_weight/std": 0.0001656018060285309, "train/tensor_grad_model_layers_7_input_layernorm_weight/max_abs": 0.00095367431640625, "train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/norm": 0.7546016667998484, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/mean": -2.0142178982496257e-06, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/std": 0.0008511400353156234, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/max_abs": 0.005828857421875, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/norm": 0.7079498898720334, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/mean": -3.5921111702919006e-06, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/std": 0.0007984289616214263, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/max_abs": 0.00775146484375, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/norm": 0.014999692337536678, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/mean": 7.618218660354614e-06, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/std": 0.00033254078012595953, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/max_abs": 0.001953125, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/norm": 0.36997189853249035, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/mean": 7.601454854011536e-06, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/std": 0.0007226019432821048, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/max_abs": 0.0052490234375, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/norm": 0.35389613122347957, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/mean": 2.362998202443123e-06, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/std": 0.0006912039625101131, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/max_abs": 0.006256103515625, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/norm": 0.002265625107904958, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/mean": -1.55740735863219e-08, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/std": 4.425062929217393e-06, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/max_abs": 6.4849853515625e-05, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/norm": 0.002163481600807979, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/mean": -6.673508323729038e-08, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/std": 4.221860208729979e-06, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/max_abs": 3.981590270996094e-05, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_6_input_layernorm_weight/norm": 0.007680140530203822, "train/tensor_grad_model_layers_6_input_layernorm_weight/mean": 1.737847924232483e-05, "train/tensor_grad_model_layers_6_input_layernorm_weight/std": 0.00016914082204828, "train/tensor_grad_model_layers_6_input_layernorm_weight/max_abs": 0.000736236572265625, "train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/norm": 0.7572818746725153, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/mean": -4.3795444071292877e-07, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/std": 0.0008547169237887492, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/max_abs": 0.007049560546875, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/norm": 0.7875399867002663, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/mean": -3.2745301723480225e-06, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/std": 0.0008873731221887825, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/max_abs": 0.00872802734375, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/norm": 0.017560181466391808, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/mean": 3.390759229660034e-05, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/std": 0.00038804992747932545, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/max_abs": 0.00139617919921875, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/norm": 0.4385958978582016, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/mean": 2.0489096641540525e-07, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/std": 0.0008566329475932972, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/max_abs": 0.006011962890625, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/norm": 0.3936700770084772, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/mean": -4.682460712501779e-07, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/std": 0.0007688871086946725, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/max_abs": 0.00823974609375, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/norm": 0.0025617701663956053, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/mean": -3.9261067286133766e-08, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/std": 5.003466769153688e-06, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/max_abs": 6.818771362304688e-05, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/norm": 0.002631145786676072, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/mean": -3.879540599882603e-08, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/std": 5.138961310076266e-06, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/max_abs": 5.459785461425781e-05, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_5_input_layernorm_weight/norm": 0.008047597318007088, "train/tensor_grad_model_layers_5_input_layernorm_weight/mean": -3.399909473955631e-07, "train/tensor_grad_model_layers_5_input_layernorm_weight/std": 0.00017868155710299555, "train/tensor_grad_model_layers_5_input_layernorm_weight/max_abs": 0.000766754150390625, "train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/norm": 0.893070010286946, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/mean": 7.382477633655071e-07, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/std": 0.0010066488793133873, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/max_abs": 0.008056640625, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/norm": 0.9250547238220178, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/mean": -6.165355443954468e-06, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/std": 0.0010431210896587194, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/max_abs": 0.007171630859375, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/norm": 0.018183048129474702, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/mean": 1.928955316543579e-05, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/std": 0.0004031774962814704, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/max_abs": 0.00177001953125, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/norm": 0.48769862701304106, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/mean": -1.239473931491375e-06, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/std": 0.0009525365850671751, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/max_abs": 0.009765625, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/norm": 0.5041465482034263, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/mean": -4.434026777744293e-06, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/std": 0.0009851406978738572, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/max_abs": 0.007293701171875, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/norm": 0.002579598592699658, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/mean": 6.875779945403336e-08, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/std": 5.038296527592207e-06, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/max_abs": 5.078315734863281e-05, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/norm": 0.002734198734869189, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/mean": -2.3712345864623785e-08, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/std": 5.342103183918348e-06, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/max_abs": 4.982948303222656e-05, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_4_input_layernorm_weight/norm": 0.01052831013690247, "train/tensor_grad_model_layers_4_input_layernorm_weight/mean": -1.8071383237838745e-05, "train/tensor_grad_model_layers_4_input_layernorm_weight/std": 0.0002329729119698576, "train/tensor_grad_model_layers_4_input_layernorm_weight/max_abs": 0.0010833740234375, "train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/norm": 1.0505652223336508, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/mean": 2.5192275643348694e-06, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/std": 0.0011842580330185134, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/max_abs": 0.01019287109375, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/norm": 1.0447064614408066, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/mean": 2.6007983251474798e-06, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/std": 0.0011766864355587643, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/max_abs": 0.0091552734375, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/norm": 0.020683129569258243, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/mean": -6.891787052154542e-05, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/std": 0.0004537914635213569, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/max_abs": 0.00145721435546875, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/norm": 0.6772388768653869, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/mean": 5.2871182560920715e-06, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/std": 0.0013227328199398023, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/max_abs": 0.00885009765625, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/norm": 0.6356109155465426, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/mean": 2.8330832719802856e-06, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/std": 0.0012414277473263577, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/max_abs": 0.01092529296875, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/norm": 0.004203062460746068, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/mean": 1.628650352358818e-07, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/std": 8.205381582890514e-06, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/max_abs": 0.0001087188720703125, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/norm": 0.0038438257099948896, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/mean": 1.6309786587953568e-07, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/std": 7.5000117007145865e-06, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/max_abs": 6.341934204101562e-05, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_3_input_layernorm_weight/norm": 0.012680099615755315, "train/tensor_grad_model_layers_3_input_layernorm_weight/mean": -3.0443072319030762e-05, "train/tensor_grad_model_layers_3_input_layernorm_weight/std": 0.0002797351357872544, "train/tensor_grad_model_layers_3_input_layernorm_weight/max_abs": 0.00118255615234375, "train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/norm": 1.359781840161146, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/mean": -5.511566996574402e-06, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/std": 0.001533029250081645, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/max_abs": 0.01019287109375, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/norm": 1.3890934772277073, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/mean": 5.669891834259033e-06, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/std": 0.0015663875075158113, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/max_abs": 0.01214599609375, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/norm": 0.029950915753115565, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/mean": -3.460794687271118e-05, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/std": 0.0006633665069236685, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/max_abs": 0.002532958984375, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/norm": 0.8457144016280518, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/mean": 2.291053533554077e-05, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/std": 0.001651296646585802, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/max_abs": 0.01226806640625, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/norm": 0.8718143783878365, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/mean": 4.575587809085846e-06, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/std": 0.001702762793163496, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/max_abs": 0.01556396484375, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/norm": 0.003984350401320912, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/mean": -1.3563408174377402e-08, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/std": 7.781938305002603e-06, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/max_abs": 7.534027099609375e-05, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/norm": 0.004042434530392268, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/mean": -1.0107214620802552e-08, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/std": 7.895382470794292e-06, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/max_abs": 7.390975952148438e-05, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_2_input_layernorm_weight/norm": 0.01634202061296293, "train/tensor_grad_model_layers_2_input_layernorm_weight/mean": -1.6802921891212463e-05, "train/tensor_grad_model_layers_2_input_layernorm_weight/std": 0.0003619688785404763, "train/tensor_grad_model_layers_2_input_layernorm_weight/max_abs": 0.0016326904296875, "train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/norm": 1.622936405687018, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/mean": -2.0614825189113617e-06, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/std": 0.0018290869677934636, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/max_abs": 0.01165771484375, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/norm": 1.6502842976104004, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/mean": 7.2177499532699585e-06, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/std": 0.001859627934527953, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/max_abs": 0.01446533203125, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/norm": 0.03126985693910291, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/mean": 8.410215377807617e-05, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/std": 0.000688121658388342, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/max_abs": 0.0023345947265625, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/norm": 0.9643301272382381, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/mean": 8.304341463372111e-07, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/std": 0.0018834577105885449, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/max_abs": 0.01275634765625, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/norm": 1.0230935953082407, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/mean": 1.0567717254161835e-05, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/std": 0.0019982397980901007, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/max_abs": 0.0184326171875, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/norm": 0.004927389985235503, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/mean": 4.525645636022091e-09, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/std": 9.623830616616655e-06, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/max_abs": 0.00011539459228515625, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/norm": 0.005701863618059202, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/mean": -1.7128513718489558e-08, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/std": 1.1136478952345827e-05, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/max_abs": 0.00010251998901367188, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_1_input_layernorm_weight/norm": 0.01918604483196558, "train/tensor_grad_model_layers_1_input_layernorm_weight/mean": 9.324867278337479e-06, "train/tensor_grad_model_layers_1_input_layernorm_weight/std": 0.0004257842314802969, "train/tensor_grad_model_layers_1_input_layernorm_weight/max_abs": 0.00182342529296875, "train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/norm": 2.129669444622657, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/mean": -2.535292878746986e-06, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/std": 0.002400357431670027, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/max_abs": 0.0174560546875, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/norm": 2.2479543923743894, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/mean": -7.683411240577698e-09, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/std": 0.0025358019538881085, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/max_abs": 0.0224609375, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/norm": 0.047971691451455564, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/mean": -6.477679562522098e-06, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/std": 0.001065419855838082, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/max_abs": 0.005157470703125, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/norm": 2.037830199650457, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/mean": 9.721261449158192e-07, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/std": 0.003980138465929412, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/max_abs": 0.037353515625, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/norm": 1.974809313876121, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/mean": 7.370021194219589e-06, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/std": 0.0038570515437090767, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/max_abs": 0.0294189453125, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/norm": 0.01663034412528906, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/mean": 3.700843080878258e-07, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/std": 3.248117562247807e-05, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/max_abs": 0.00030517578125, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/norm": 0.01809298647783285, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/mean": -4.034372977912426e-07, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/std": 3.5337946708383607e-05, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/max_abs": 0.000637054443359375, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_layers_0_input_layernorm_weight/norm": 0.03457925756213994, "train/tensor_grad_model_layers_0_input_layernorm_weight/mean": -7.304549217224121e-05, "train/tensor_grad_model_layers_0_input_layernorm_weight/std": 0.0007638811556086561, "train/tensor_grad_model_layers_0_input_layernorm_weight/max_abs": 0.0031890869140625, "train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_grad_model_embed_tokens_weight/norm": 3.6993218702628035, "train/tensor_grad_model_embed_tokens_weight/mean": 5.578622221946716e-06, "train/tensor_grad_model_embed_tokens_weight/std": 0.0012787561777355767, "train/tensor_grad_model_embed_tokens_weight/max_abs": 0.185546875, "train/tensor_grad_model_embed_tokens_weight/frac_near_dtype_limit": 0.0, "train/tensor_grad_model_embed_tokens_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_embed_tokens_weight/norm": 14.4375, "train/tensor_param_model_embed_tokens_weight/mean": 4.4345855712890625e-05, "train/tensor_param_model_embed_tokens_weight/std": 0.02001953125, "train/tensor_param_model_embed_tokens_weight/max_abs": 0.0966796875, "train/tensor_param_model_embed_tokens_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_embed_tokens_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/mean": -0.0002288818359375, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/max_abs": 0.083984375, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/mean": 0.00017547607421875, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/max_abs": 0.08349609375, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/mean": 0.00013065338134765625, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/max_abs": 0.0830078125, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/mean": -3.933906555175781e-05, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/max_abs": 0.08642578125, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_0_mlp_up_proj_weight/mean": -3.5762786865234375e-05, "train/tensor_param_model_layers_0_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_0_mlp_up_proj_weight/max_abs": 0.08984375, "train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_0_mlp_down_proj_weight/mean": 6.67572021484375e-05, "train/tensor_param_model_layers_0_mlp_down_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_0_mlp_down_proj_weight/max_abs": 0.08642578125, "train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_0_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_0_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_0_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/mean": 3.3855438232421875e-05, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/max_abs": 0.08447265625, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/mean": -2.8014183044433594e-05, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/max_abs": 0.08837890625, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/mean": -0.00020503997802734375, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/max_abs": 0.08056640625, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/mean": 3.4123659133911133e-06, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/max_abs": 0.076171875, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_1_mlp_up_proj_weight/mean": -6.67572021484375e-05, "train/tensor_param_model_layers_1_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_1_mlp_up_proj_weight/max_abs": 0.080078125, "train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_1_mlp_down_proj_weight/mean": -5.269050598144531e-05, "train/tensor_param_model_layers_1_mlp_down_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_1_mlp_down_proj_weight/max_abs": 0.07958984375, "train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_1_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_1_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_1_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/mean": -0.00015735626220703125, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/max_abs": 0.0869140625, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/mean": -3.427267074584961e-06, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/max_abs": 0.08349609375, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/mean": 2.8848648071289062e-05, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/max_abs": 0.08544921875, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/mean": -4.696846008300781e-05, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/max_abs": 0.08447265625, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_2_mlp_up_proj_weight/mean": 3.814697265625e-05, "train/tensor_param_model_layers_2_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_2_mlp_up_proj_weight/max_abs": 0.080078125, "train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_2_mlp_down_proj_weight/mean": -0.00014400482177734375, "train/tensor_param_model_layers_2_mlp_down_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_2_mlp_down_proj_weight/max_abs": 0.09130859375, "train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_2_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_2_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_2_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/mean": -9.679794311523438e-05, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/max_abs": 0.0830078125, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/mean": 6.961822509765625e-05, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/max_abs": 0.0810546875, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/mean": 1.8924474716186523e-06, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/max_abs": 0.080078125, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/mean": -0.0002079010009765625, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/max_abs": 0.07958984375, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_3_mlp_up_proj_weight/mean": -0.00010156631469726562, "train/tensor_param_model_layers_3_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_3_mlp_up_proj_weight/max_abs": 0.083984375, "train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_3_mlp_down_proj_weight/mean": -0.00019931793212890625, "train/tensor_param_model_layers_3_mlp_down_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_3_mlp_down_proj_weight/max_abs": 0.08544921875, "train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_3_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_3_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_3_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/mean": -8.20159912109375e-05, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/max_abs": 0.0869140625, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/mean": 5.745887756347656e-05, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/max_abs": 0.078125, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/norm": 2.546875, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/mean": -8.20159912109375e-05, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/std": 0.0198974609375, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/max_abs": 0.0810546875, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/mean": -9.72747802734375e-05, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/max_abs": 0.091796875, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_4_mlp_up_proj_weight/mean": -1.704692840576172e-05, "train/tensor_param_model_layers_4_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_4_mlp_up_proj_weight/max_abs": 0.08740234375, "train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_4_mlp_down_proj_weight/mean": 4.363059997558594e-05, "train/tensor_param_model_layers_4_mlp_down_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_4_mlp_down_proj_weight/max_abs": 0.0888671875, "train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_4_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_4_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_4_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/mean": -0.00010919570922851562, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/max_abs": 0.08056640625, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/mean": -6.29425048828125e-05, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/max_abs": 0.0859375, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/norm": 2.53125, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/mean": -0.0001392364501953125, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/std": 0.019775390625, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/max_abs": 0.09033203125, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/mean": 0.00015544891357421875, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/max_abs": 0.0908203125, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_5_mlp_up_proj_weight/mean": -0.00013446807861328125, "train/tensor_param_model_layers_5_mlp_up_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_5_mlp_up_proj_weight/max_abs": 0.09033203125, "train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_mlp_down_proj_weight/norm": 4.40625, "train/tensor_param_model_layers_5_mlp_down_proj_weight/mean": 0.0001678466796875, "train/tensor_param_model_layers_5_mlp_down_proj_weight/std": 0.0198974609375, "train/tensor_param_model_layers_5_mlp_down_proj_weight/max_abs": 0.087890625, "train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_5_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_5_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_5_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/norm": 2.53125, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/mean": 1.7523765563964844e-05, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/std": 0.019775390625, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/max_abs": 0.08349609375, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/mean": 0.000164031982421875, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/max_abs": 0.07958984375, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/mean": -0.0002498626708984375, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/max_abs": 0.0791015625, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/norm": 2.546875, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/mean": 0.0002117156982421875, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/std": 0.0198974609375, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/max_abs": 0.0791015625, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_6_mlp_up_proj_weight/mean": 8.869171142578125e-05, "train/tensor_param_model_layers_6_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_6_mlp_up_proj_weight/max_abs": 0.08056640625, "train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_6_mlp_down_proj_weight/mean": 0.00010395050048828125, "train/tensor_param_model_layers_6_mlp_down_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_6_mlp_down_proj_weight/max_abs": 0.0830078125, "train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_6_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_6_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_6_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/mean": -0.0002689361572265625, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/max_abs": 0.07861328125, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/mean": 0.00031280517578125, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/max_abs": 0.08056640625, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/mean": -0.00011968612670898438, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/max_abs": 0.08251953125, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/norm": 2.578125, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/mean": -8.344650268554688e-05, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/std": 0.0201416015625, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/max_abs": 0.078125, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_7_mlp_up_proj_weight/mean": -8.296966552734375e-05, "train/tensor_param_model_layers_7_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_7_mlp_up_proj_weight/max_abs": 0.0830078125, "train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_mlp_down_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_7_mlp_down_proj_weight/mean": 2.86102294921875e-05, "train/tensor_param_model_layers_7_mlp_down_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_7_mlp_down_proj_weight/max_abs": 0.09033203125, "train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_7_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_7_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_7_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/norm": 2.546875, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/mean": 9.417533874511719e-06, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/std": 0.0198974609375, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/max_abs": 0.08154296875, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/mean": 8.678436279296875e-05, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/max_abs": 0.0810546875, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/norm": 2.53125, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/mean": -0.0002593994140625, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/std": 0.019775390625, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/max_abs": 0.07958984375, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/norm": 2.5625, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/mean": -1.2278556823730469e-05, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/max_abs": 0.07666015625, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_mlp_up_proj_weight/norm": 4.4375, "train/tensor_param_model_layers_8_mlp_up_proj_weight/mean": -0.00018215179443359375, "train/tensor_param_model_layers_8_mlp_up_proj_weight/std": 0.02001953125, "train/tensor_param_model_layers_8_mlp_up_proj_weight/max_abs": 0.1025390625, "train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_mlp_down_proj_weight/norm": 4.40625, "train/tensor_param_model_layers_8_mlp_down_proj_weight/mean": -0.00020503997802734375, "train/tensor_param_model_layers_8_mlp_down_proj_weight/std": 0.0198974609375, "train/tensor_param_model_layers_8_mlp_down_proj_weight/max_abs": 0.0830078125, "train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_input_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_8_input_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_8_input_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_8_input_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/norm": 11.3125, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/mean": 1.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/std": 0.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/max_abs": 1.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit": 0.0, "train/tensor_param_model_norm_weight/norm": 11.3125, "train/tensor_param_model_norm_weight/mean": 1.0, "train/tensor_param_model_norm_weight/std": 0.0, "train/tensor_param_model_norm_weight/max_abs": 1.0, "train/tensor_param_model_norm_weight/frac_near_dtype_limit": 0.0, "train/tensor_param_model_norm_weight/frac_near_user_limit": 0.0} +{"step": 40, "epoch": 0.053926525109538256, "timestamp": 1786256473.940319, "loss": 102.0810546875, "grad_norm": 39.5, "learning_rate": 0.0005, "train/total_time_seconds": 42.01612686738372, "train/time_per_step_avg": 1.050403171684593, "train/epoch_time_elapsed": 81.34817474707961, "train/estimated_remaining_minutes": 12.42977086493435} diff --git a/outio/sweep_summary.json b/outio/sweep_summary.json new file mode 100644 index 0000000000000000000000000000000000000000..b8c1afdf90447072fda5e3b1128a53900b0a4e49 --- /dev/null +++ b/outio/sweep_summary.json @@ -0,0 +1,8 @@ +[ + { + "variant": "mlp-linear-9L", + "eval_loss": 3.108103036880493, + "out": "outio/mlp-linear-9L_run", + "run_name": "LM-mlp-linear-9L-2.0M-20260809-055343" + } +] \ No newline at end of file diff --git a/sweep.py b/sweep.py new file mode 100644 index 0000000000000000000000000000000000000000..f28c029c1991642fd523af104bca03b370fada6f --- /dev/null +++ b/sweep.py @@ -0,0 +1,161 @@ +#!/usr/bin/env python3 +"""Sweep explicit GLU / MLP variants with identical data and hyperparameters.""" +import argparse +import copy +import json +import re +import time +import os +from pathlib import Path + +import yaml +import wandb +import torch +from transformers import AutoTokenizer, set_seed +from exp import TinyLlamaConfig, TinyLlamaForCausalLM, build_dataset, create_trainer + + +def format_param_count(total_params: int) -> str: + """Return human‑readable string with M or B suffix, 1 decimal.""" + if total_params >= 1e9: + return f"{total_params / 1e9:.1f}B" + else: + return f"{total_params / 1e6:.1f}M" + + +def parse_variant(variant: str): + """ + Parse variant string into (prefix, activation, layers). + Formats: + glu-silu -> ('glu', 'silu', None) + mlp-s10-10L -> ('mlp', 's10', 10) + glu-relu-8L -> ('glu', 'relu', 8) + """ + # Pattern: optional layers at end with 'L' + match = re.fullmatch(r'(glu|mlp)-([a-zA-Z0-9]+)(?:-(\d+)L)?', variant) + if not match: + raise ValueError( + f"Invalid variant format: '{variant}'. " + "Expected: -[-L] e.g. glu-silu-10L" + ) + prefix, act, layers_str = match.groups() + layers = int(layers_str) if layers_str is not None else None + return prefix, act, layers + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--config", required=True, help="Base YAML config") + parser.add_argument( + "--variants", + nargs="+", + required=True, + help="List of variants: e.g. glu-silu-10L mlp-relu-8L" + ) + parser.add_argument("--push", action="store_true") + args = parser.parse_args() + + with open(args.config) as f: + base = yaml.safe_load(f) + + seed = base.get("training", {}).get("seed", 42) + set_seed(seed) + + tok_name = base["model"].get("tokenizer_name", "meta-llama/Llama-2-7b-hf") + tokenizer = AutoTokenizer.from_pretrained(tok_name) + if tokenizer.pad_token is None: + tokenizer.pad_token = tokenizer.eos_token + + msl = base["model"].get("max_position_embeddings", 512) + train_ds = build_dataset( + tokenizer, + max_seq_len=msl, + split="train", + max_samples=None # limit to 50k samples + ) + eval_ds = build_dataset( + tokenizer, + max_seq_len=msl, + split="validation", + max_samples=None # no limit for eval + ) + + results = [] + + for variant in args.variants: + prefix, act, layers = parse_variant(variant) + + # Validation: mlp-situglu and mlp-waleed are banned + if prefix == "mlp" and act in ("situglu", "waleed"): + raise ValueError( + f"Activation '{act}' requires a gated architecture (GLU). " + f"Please use 'glu-{act}' instead." + ) + + # Build config overrides + cfg = copy.deepcopy(base) + cfg["model"]["mlp_type"] = prefix + cfg["model"]["activation"] = act + if layers is not None: + cfg["model"]["num_hidden_layers"] = layers + + # Create a descriptive suffix for folders / run names + layer_suffix = f"-{layers}L" if layers is not None else "" + variant_label = f"{prefix}-{act}{layer_suffix}" + + # Unique output directory + out_dir = Path(cfg["training"]["output_dir"]).parent / f"{variant_label}_run" + cfg["training"]["output_dir"] = str(out_dir) + + # Re-seed for reproducibility across variants + set_seed(seed) + + print(f"\n{'='*60}\n>>> Variant: {variant_label} | Out: {out_dir}\n{'='*60}") + + # Instantiate model – no explicit _attn_implementation (use default) + config = TinyLlamaConfig(**cfg["model"]) + model = TinyLlamaForCausalLM(config) + # Cast to bfloat16 if desired (we keep the same as before) + model = model.to(torch.bfloat16) + + total_params = sum(p.numel() for p in model.parameters()) + param_str = format_param_count(total_params) + timestamp = time.strftime("%Y%m%d-%H%M%S") + run_name = f"LM-{variant_label}-{param_str}-{timestamp}" + cfg["training"]["run_name"] = run_name + + # Also update hub_model_id to include variant and layers + hub_id_base = cfg["training"].get("hub_model_id", "tiny-llama-lab") + cfg["training"]["hub_model_id"] = f"{hub_id_base}-{variant_label}" + + # Ensure a fresh WandB run – remove any global WANDB_RUN_ID + os.environ.pop("WANDB_RUN_ID", None) + + trainer = create_trainer(model, tokenizer, cfg, train_ds, eval_ds) + + try: + trainer.train() + metrics = trainer.evaluate() + results.append({ + "variant": variant_label, + "eval_loss": metrics.get("eval_loss"), + "out": str(out_dir), + "run_name": run_name, + }) + trainer.save_model(str(out_dir)) + if args.push or cfg["training"].get("push_to_hub", False): + trainer.push_to_hub() + finally: + # Explicitly finish WandB run to avoid re‑using the same run + wandb.finish() + + # Save summary + summary = Path(base["training"]["output_dir"]).parent / "sweep_summary.json" + summary.write_text(json.dumps(results, indent=2)) + print("\nSweep complete:") + for r in results: + print(f" {r['variant']:20s} eval_loss={r['eval_loss']:.4f}") + + +if __name__ == "__main__": + main() diff --git a/train.py b/train.py new file mode 100644 index 0000000000000000000000000000000000000000..fffe20792a55e2089dc55e4f1262c894b03bd41b --- /dev/null +++ b/train.py @@ -0,0 +1,70 @@ +#!/usr/bin/env python3 +"""Train one TinyLlama variant from a YAML config.""" +import argparse +import yaml +import torch + +from transformers import AutoTokenizer, set_seed +from exp import TinyLlamaConfig, TinyLlamaForCausalLM, build_dataset, create_trainer + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--config", required=True, help="Path to YAML config") + parser.add_argument("--push", action="store_true", help="Push final model to HF Hub") + args = parser.parse_args() + + with open(args.config) as f: + cfg = yaml.safe_load(f) + + # Explicit seed before any randomness + seed = cfg.get("training", {}).get("seed", 42) + set_seed(seed) + + model_cfg = cfg["model"] + train_cfg = cfg.get("training", {}) + + # Tokenizer + tok_name = model_cfg.pop("tokenizer_name", "meta-llama/Llama-2-7b-hf") + tokenizer = AutoTokenizer.from_pretrained(tok_name) + if tokenizer.pad_token is None: + tokenizer.pad_token = tokenizer.eos_token + + # Model – the config must contain mlp_type and activation + tiny_config = TinyLlamaConfig(**model_cfg) + # No explicit attention implementation – let transformers pick the default + model = TinyLlamaForCausalLM(tiny_config) + model = model.to(torch.bfloat16) + + n_params = sum(p.numel() for p in model.parameters()) / 1e6 + print(f"Model: {n_params:.2f}M params | MLP type: {tiny_config.mlp_type} | Activation: {tiny_config.activation}") + + # Data + msl = model_cfg.get("max_position_embeddings", 512) + train_ds = build_dataset( + tokenizer, + max_seq_len=msl, + split="train", + max_samples=None # limit to 50k samples + ) + eval_ds = build_dataset( + tokenizer, + max_seq_len=msl, + split="validation", + max_samples=None # no limit for eval + ) + + # Train + trainer = create_trainer(model, tokenizer, cfg, train_ds, eval_ds) + trainer.train() + + # Save & push + out = train_cfg.get("output_dir", "./out") + trainer.save_model(out) + if args.push or train_cfg.get("push_to_hub", False): + trainer.push_to_hub() + print(f"Done. Artifacts in {out}") + + +if __name__ == "__main__": + main() diff --git a/wandb/debug-internal.log b/wandb/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..04886cdd8d75e6ef4a4955a1434a422418e44d50 --- /dev/null +++ b/wandb/debug-internal.log @@ -0,0 +1,40 @@ +{"time":"2026-08-09T07:02:13.954953014Z","level":"INFO","msg":"wandb-core"} +{"time":"2026-08-09T07:02:13.955105829Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"} +{"time":"2026-08-09T07:02:14.215123169Z","level":"INFO","msg":"stream: created new stream","id":"uvqyddz0"} +{"time":"2026-08-09T07:02:14.215184619Z","level":"INFO","msg":"handler: started"} +{"time":"2026-08-09T07:02:14.215266968Z","level":"INFO","msg":"stream: started"} +{"time":"2026-08-09T07:02:14.215275401Z","level":"INFO","msg":"writer: started","stream_id":"uvqyddz0"} +{"time":"2026-08-09T07:02:14.215296287Z","level":"INFO","msg":"sender: started"} +{"time":"2026-08-09T07:02:15.114109279Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1} +{"time":"2026-08-09T07:02:15.211087322Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:02:30.114358029Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":0,"events_lines":2,"console_offset":1,"console_lines":4,"uploaded_len":2} +{"time":"2026-08-09T07:02:30.230090173Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:02:45.11459263Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":2,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T07:02:45.228934504Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:02:56.598899602Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":737} +{"time":"2026-08-09T07:02:56.598941378Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1} +{"time":"2026-08-09T07:02:56.606194892Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":3095} +{"time":"2026-08-09T07:02:56.606379301Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":14} +{"time":"2026-08-09T07:02:56.610759144Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":4507} +{"time":"2026-08-09T07:02:56.610912588Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":13} +{"time":"2026-08-09T07:02:56.613175719Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":5301} +{"time":"2026-08-09T07:02:56.618404048Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1932} +{"time":"2026-08-09T07:02:56.629287315Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":10255} +{"time":"2026-08-09T07:02:56.629322845Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1} +{"time":"2026-08-09T07:02:56.63322527Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":11868} +{"time":"2026-08-09T07:02:56.633338533Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":10} +{"time":"2026-08-09T07:02:56.636529541Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":13176} +{"time":"2026-08-09T07:02:56.63797994Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":477} +{"time":"2026-08-09T07:02:56.643113777Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":14979} +{"time":"2026-08-09T07:02:56.643188638Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":12} +{"time":"2026-08-09T07:03:00.155597928Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":1,"events_offset":4,"events_lines":2,"console_offset":4,"console_lines":2} +{"time":"2026-08-09T07:03:01.144833138Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:03:15.114830048Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":6,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T07:03:15.243621253Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:03:30.114703175Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":8,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T07:03:30.219274044Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:03:40.29903607Z","level":"ERROR","msg":"runupserter: failed to upload changes","error":"POST https://api.wandb.ai/graphql giving up after 1 attempt(s): context canceled"} +{"time":"2026-08-09T07:03:40.30029433Z","level":"ERROR","msg":"runfiles: CreateRunFiles returned error: POST https://api.wandb.ai/graphql giving up after 1 attempt(s): context canceled"} +{"time":"2026-08-09T07:03:40.300493766Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"} +{"time":"2026-08-09T07:03:40.322418888Z","level":"INFO","msg":"filestream: sending request","total_files":3,"history_offset":1,"history_lines":1,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T07:03:40.3225757Z","level":"ERROR+4","msg":"filestream: fatal error: filestream: error making HTTP request: POST https://api.wandb.ai/files/deepnevro-deepnevro/huggingface/uvqyddz0/file_stream giving up after 1 attempt(s): context canceled. got response: "} diff --git a/wandb/debug.log b/wandb/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..6b23e39286dbd88cb2f9a763f55c9b5f9e2fb13b --- /dev/null +++ b/wandb/debug.log @@ -0,0 +1,27 @@ +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_setup.py:_flush():81] Current SDK version is 0.28.1 +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_setup.py:_flush():81] Configure stats pid to 474098 +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_setup.py:_flush():81] Loading settings from environment variables +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_070213-uvqyddz0/logs/debug.log +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_070213-uvqyddz0/logs/debug-internal.log +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_init.py:init():772] calling init triggers +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_init.py:init():777] wandb.init called with sweep_config: {} +config: {'_wandb': {}} +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_init.py:init():820] starting backend +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_init.py:init():826] Connected to an existing wandb-core service via WANDB_SERVICE +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_init.py:init():835] sending inform_init request +2026-08-09 07:02:14,215 INFO MainThread:474098 [wandb_init.py:init():840] backend started and connected +2026-08-09 07:02:14,218 INFO MainThread:474098 [wandb_init.py:init():910] updated telemetry +2026-08-09 07:02:14,225 INFO MainThread:474098 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout +2026-08-09 07:02:14,452 INFO MainThread:474098 [wandb_init.py:init():978] starting run threads in backend +2026-08-09 07:02:14,525 INFO MainThread:474098 [wandb_run.py:_console_start():2621] atexit reg +2026-08-09 07:02:14,525 INFO MainThread:474098 [wandb_run.py:_redirect():2471] redirect: wrap_raw +2026-08-09 07:02:14,525 INFO MainThread:474098 [wandb_run.py:_redirect():2540] Wrapping output streams. +2026-08-09 07:02:14,525 INFO MainThread:474098 [wandb_run.py:_redirect():2563] Redirects installed. +2026-08-09 07:02:14,528 INFO MainThread:474098 [wandb_init.py:init():1016] run started, returning control to user process +2026-08-09 07:02:14,529 INFO MainThread:474098 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.15.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 94, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'glu', 'activation': 'tanh', 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/glu-tanh-94L_run', 'per_device_train_batch_size': 128, 'num_train_epochs': 1, 'max_steps': 1500, 'learning_rate': 0.001, 'lr_scheduler_type': 'constant', 'lr_scheduler_kwargs': None, 'warmup_steps': 0, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.01, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 4, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-glu-tanh-94L-15.9M-20260809-070212', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 50, 'eval_delay': 0, 'per_device_eval_batch_size': 128, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': True, 'hub_token': '', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/A-glu-tanh-94L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1} +2026-08-09 07:02:14,532 INFO MainThread:474098 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 15949440 - > +2026-08-09 07:02:14,532 INFO MainThread:474098 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 15949440 None +2026-08-09 07:03:39,932 INFO MainThread:474098 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/huggingface/uvqyddz0 +2026-08-09 07:03:39,933 INFO MainThread:474098 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0 +2026-08-09 07:03:39,933 INFO MainThread:474098 [wandb_run.py:_restore():2570] restore +2026-08-09 07:03:39,933 INFO MainThread:474098 [wandb_run.py:_restore():2576] restore done diff --git a/wandb/run-20260809_035819-cvzjg5ej/files/config.yaml b/wandb/run-20260809_035819-cvzjg5ej/files/config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..247b6a13773dafecbde0bc870e6d159065b3c337 --- /dev/null +++ b/wandb/run-20260809_035819-cvzjg5ej/files/config.yaml @@ -0,0 +1,433 @@ +_name_or_path: + value: "" +_wandb: + value: + cli_version: 0.28.1 + e: + oz75ofyjp8ovu1ftttzp3wyx4l4oquya: + args: + - --config + - configs/baseline.yaml + - --variants + - mlp-s10-94L + - --push + codePath: sweep.py + codePathLocal: sweep.py + cpu_count: 112 + cpu_count_logical: 224 + cudaVersion: "12.4" + disk: + /: + total: "1560765693952" + used: "708205846528" + email: deepnevro@gmail.com + executable: /mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python + git: + commit: 34b8d2e8f9a0c5751333310e69fa0c1056381deb + remote: https://github.com/deepnevro/Activation.git + gpu: NVIDIA H100 80GB HBM3 + gpu_count: 8 + gpu_nvidia: + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-39c684a5-fde6-83d7-1663-0859795881ae + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-132944c4-b689-2b5f-89a4-d730401677ab + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bc6c3e3c-9b90-09ca-c034-774961847c54 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea + host: deeplens-k3s-node1 + memory: + total: "2164089937920" + os: Linux-5.15.0-126-generic-x86_64-with-glibc2.35 + program: /mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py + python: CPython 3.11.15 + root: /mnt/data/zainulabideen/zain-exp/notebooks/Activation + startedAt: "2026-08-09T03:58:19.280139Z" + writerId: oz75ofyjp8ovu1ftttzp3wyx4l4oquya + m: + - "1": train/global_step + "6": + - 3 + "7": [] + - "2": '*' + "5": 1 + "6": + - 1 + "7": [] + python_version: 3.11.15 + t: + "1": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "2": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "3": + - 2 + - 7 + - 13 + - 19 + - 41 + - 62 + - 66 + "4": 3.11.15 + "5": 0.28.1 + "6": 5.15.0.dev0 + "9": + "1": transformers_trainer + "12": 0.28.1 + "13": linux-x86_64 +accelerator_config: + value: + dispatch_batches: null + even_batches: true + gradient_accumulation_kwargs: null + non_blocking: false + split_batches: false + use_seedable_sampler: true +activation: + value: s10 +adam_beta1: + value: 0.9 +adam_beta2: + value: 0.999 +adam_epsilon: + value: 1e-08 +architectures: + value: null +attention_bias: + value: false +attention_dropout: + value: 0 +auto_find_batch_size: + value: false +average_tokens_across_devices: + value: true +batch_eval_metrics: + value: false +bf16: + value: true +bf16_full_eval: + value: false +bos_token_id: + value: 1 +chunk_size_feed_forward: + value: 0 +data_seed: + value: 42 +dataloader_drop_last: + value: false +dataloader_in_order: + value: true +dataloader_multiprocessing_context: + value: null +dataloader_num_workers: + value: 0 +dataloader_persistent_workers: + value: false +dataloader_pin_memory: + value: true +dataloader_prefetch_factor: + value: null +ddp_backend: + value: null +ddp_broadcast_buffers: + value: null +ddp_bucket_cap_mb: + value: null +ddp_find_unused_parameters: + value: null +ddp_static_graph: + value: null +ddp_timeout: + value: 1800 +debug: + value: [] +deepspeed: + value: null +disable_tqdm: + value: false +do_eval: + value: true +do_predict: + value: false +do_train: + value: false +dtype: + value: null +enable_jit_checkpoint: + value: false +eos_token_id: + value: 2 +eval_accumulation_steps: + value: null +eval_delay: + value: 0 +eval_do_concat_batches: + value: true +eval_on_start: + value: false +eval_steps: + value: 50 +eval_strategy: + value: steps +eval_use_gather_object: + value: false +fp16: + value: false +fp16_full_eval: + value: false +fsdp: + value: null +fsdp_config: + value: null +full_determinism: + value: false +gradient_accumulation_steps: + value: 4 +gradient_checkpointing: + value: false +gradient_checkpointing_kwargs: + value: null +greater_is_better: + value: null +head_dim: + value: 32 +hidden_act: + value: silu +hidden_size: + value: 128 +hub_always_push: + value: false +hub_model_id: + value: w-ahmad/A-mlp-s10-94L +hub_private_repo: + value: null +hub_revision: + value: null +hub_strategy: + value: every_save +hub_token: + value: +id2label: + value: + "0": LABEL_0 + "1": LABEL_1 +ignore_data_skip: + value: false +include_for_metrics: + value: [] +include_num_input_tokens_seen: + value: "no" +initializer_range: + value: 0.02 +intermediate_size: + value: 256 +is_encoder_decoder: + value: false +label_names: + value: null +label_smoothing_factor: + value: 0 +label2id: + value: + LABEL_0: 0 + LABEL_1: 1 +learning_rate: + value: 0.001 +length_column_name: + value: length +liger_kernel_config: + value: null +load_best_model_at_end: + value: false +local_rank: + value: -1 +log_level: + value: passive +log_level_replica: + value: warning +log_on_each_node: + value: true +logging_first_step: + value: false +logging_nan_inf_filter: + value: true +logging_steps: + value: 20 +logging_strategy: + value: steps +lr_scheduler_kwargs: + value: null +lr_scheduler_type: + value: constant +max_grad_norm: + value: 1 +max_position_embeddings: + value: 512 +max_steps: + value: 1500 +metric_for_best_model: + value: null +mlp_bias: + value: false +mlp_type: + value: mlp +model/num_parameters: + value: 15949440 +model_type: + value: tiny_llama +neftune_noise_alpha: + value: null +num_attention_heads: + value: 4 +num_hidden_layers: + value: 94 +num_key_value_heads: + value: 4 +num_train_epochs: + value: 1 +optim: + value: adamw_torch_fused +optim_args: + value: null +optim_target_modules: + value: null +output_attentions: + value: false +output_dir: + value: out/mlp-s10-94L_run +output_hidden_states: + value: false +pad_token_id: + value: 0 +parallelism_config: + value: null +per_device_eval_batch_size: + value: 128 +per_device_train_batch_size: + value: 128 +prediction_loss_only: + value: false +pretraining_tp: + value: 1 +problem_type: + value: null +project: + value: huggingface +push_to_hub: + value: true +remove_unused_columns: + value: false +report_to: + value: + - wandb +restore_callback_states_from_checkpoint: + value: false +resume_from_checkpoint: + value: null +return_dict: + value: true +rms_norm_eps: + value: 1e-06 +rope_parameters: + value: + rope_theta: 10000 + rope_type: default +run_name: + value: LM-mlp-s10-94L-15.9M-20260809-035818 +save_on_each_node: + value: false +save_only_model: + value: false +save_steps: + value: 100 +save_strategy: + value: steps +save_total_limit: + value: null +seed: + value: 42 +skip_memory_metrics: + value: true +tf32: + value: null +tie_word_embeddings: + value: true +tokenizer_name: + value: w-ahmad/tiny-stories-tokenizer +torch_compile: + value: false +torch_compile_backend: + value: null +torch_compile_mode: + value: null +torch_empty_cache_steps: + value: null +trackio_bucket_id: + value: null +trackio_space_id: + value: null +trackio_static_space_id: + value: null +train_sampling_strategy: + value: random +transformers_version: + value: 5.15.0.dev0 +use_cache: + value: false +use_cpu: + value: false +use_liger_kernel: + value: false +vocab_size: + value: 4096 +warmup_steps: + value: 0 +weight_decay: + value: 0.01 diff --git a/wandb/run-20260809_035819-cvzjg5ej/files/output.log b/wandb/run-20260809_035819-cvzjg5ej/files/output.log new file mode 100644 index 0000000000000000000000000000000000000000..f0041e7d14297d604a18b191b8913f9aa53a39ff --- /dev/null +++ b/wandb/run-20260809_035819-cvzjg5ej/files/output.log @@ -0,0 +1,221 @@ +[transformers] `use_return_dict` is deprecated! Use `return_dict` instead! +[INFO] Causal mask (float with -inf) applied to all attention layers. +/mnt/data/zainulabideen/zain-exp/notebooks/Activation/exp.py:487: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1831.) + "std": tensor.std().item(), + 7%|██▌ | 100/1500 [03:58<47:48, 2.05s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '28.57', 'grad_norm': '3.75', 'learning_rate': '0.001', 'epoch': '0.01078', 'train/total_time_seconds': '35.3', 'train/time_per_step_avg': '1.765', 'train/epoch_time_elapsed': '43.74', 'train/estimated_remaining_minutes': '43.53', 'train/global/act/norm': '8.829e+04', 'train/global/act/mean': '0.01651', 'train/global/act/std': '0.4295', 'train/global/act/max_abs': '8.337', 'train/global/act/frac_near_dtype_limit': '0', 'train/global/act/frac_near_user_limit': '0', 'train/global/grad/norm': '5.516', 'train/global/grad/mean': '-3.54e-06', 'train/global/grad/std': '0.0006905', 'train/global/grad/max_abs': '0.1084', 'train/global/grad/frac_near_dtype_limit': '0', 'train/global/grad/frac_near_user_limit': '0', 'train/global/param/norm': '174.8', 'train/global/param/mean': '0.001534', 'train/global/param/std': '0.04376', 'train/global/param/max_abs': '1', 'train/global/param/frac_near_dtype_limit': '0', 'train/global/param/frac_near_user_limit': '0', 'train/layer_model_layers_27/act/norm': '8942', 'train/layer_model_layers_27/act/mean': '0.018', 'train/layer_model_layers_27/act/std': '0.4279', 'train/layer_model_layers_27/act/max_abs': '5.094', 'train/layer_model_layers_27/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_27/act/frac_near_user_limit': '0', 'train/layer_model_layers_27/grad/norm': '0.3325', 'train/layer_model_layers_27/grad/mean': '-1.208e-05', 'train/layer_model_layers_27/grad/std': '0.0004102', 'train/layer_model_layers_27/grad/max_abs': '0.008545', 'train/layer_model_layers_27/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_27/grad/frac_near_user_limit': '0', 'train/layer_model_layers_87/act/norm': '9222', 'train/layer_model_layers_87/act/mean': '0.01927', 'train/layer_model_layers_87/act/std': '0.4411', 'train/layer_model_layers_87/act/max_abs': '4.281', 'train/layer_model_layers_87/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_87/act/frac_near_user_limit': '0', 'train/layer_model_layers_87/grad/norm': '0.2037', 'train/layer_model_layers_87/grad/mean': '-1.154e-06', 'train/layer_model_layers_87/grad/std': '0.0002515', 'train/layer_model_layers_87/grad/max_abs': '0.005646', 'train/layer_model_layers_87/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_87/grad/frac_near_user_limit': '0', 'train/layer_model_layers_53/act/norm': '9063', 'train/layer_model_layers_53/act/mean': '0.02554', 'train/layer_model_layers_53/act/std': '0.4333', 'train/layer_model_layers_53/act/max_abs': '4.562', 'train/layer_model_layers_53/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_53/act/frac_near_user_limit': '0', 'train/layer_model_layers_53/grad/norm': '0.2643', 'train/layer_model_layers_53/grad/mean': '-7.417e-07', 'train/layer_model_layers_53/grad/std': '0.0003263', 'train/layer_model_layers_53/grad/max_abs': '0.007202', 'train/layer_model_layers_53/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_53/grad/frac_near_user_limit': '0', 'train/layer__model_layers_31/param/norm': '17.93', 'train/layer__model_layers_31/param/mean': '0.001673', 'train/layer__model_layers_31/param/std': '0.04423', 'train/layer__model_layers_31/param/max_abs': '1', 'train/layer__model_layers_31/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_31/param/frac_near_user_limit': '0', 'train/layer_model_layers_24/act/norm': '8922', 'train/layer_model_layers_24/act/mean': '0.01278', 'train/layer_model_layers_24/act/std': '0.4268', 'train/layer_model_layers_24/act/max_abs': '5', 'train/layer_model_layers_24/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_24/act/frac_near_user_limit': '0', 'train/layer_model_layers_24/grad/norm': '0.3614', 'train/layer_model_layers_24/grad/mean': '-1.145e-05', 'train/layer_model_layers_24/grad/std': '0.000446', 'train/layer_model_layers_24/grad/max_abs': '0.007385', 'train/layer_model_layers_24/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_24/grad/frac_near_user_limit': '0', 'train/layer_model_layers_39/act/norm': '8998', 'train/layer_model_layers_39/act/mean': '0.02845', 'train/layer_model_layers_39/act/std': '0.43 +{'loss': '24.28', 'grad_norm': '0.2637', 'learning_rate': '0.001', 'epoch': '0.02157', 'train/total_time_seconds': '67.59', 'train/time_per_step_avg': '1.69', 'train/epoch_time_elapsed': '84.33', 'train/estimated_remaining_minutes': '41.11'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '5.909', 'eval_runtime': '16.38', 'eval_samples_per_second': '581.5', 'eval_steps_per_second': '4.578', 'epoch': '0.02696', 'train/total_time_seconds': '83.86', 'train/time_per_step_avg': '1.677', 'train/epoch_time_elapsed': '120.9', 'train/estimated_remaining_minutes': '40.53'} +{'loss': '23.61', 'grad_norm': '2.031', 'learning_rate': '0.001', 'epoch': '0.03235', 'train/total_time_seconds': '99.98', 'train/time_per_step_avg': '1.666', 'train/epoch_time_elapsed': '140.8', 'train/estimated_remaining_minutes': '39.99'} +{'loss': '22.8', 'grad_norm': '3.938', 'learning_rate': '0.001', 'epoch': '0.04314', 'train/total_time_seconds': '132.2', 'train/time_per_step_avg': '1.653', 'train/epoch_time_elapsed': '180.9', 'train/estimated_remaining_minutes': '39.12'} +{'loss': '22.14', 'grad_norm': '4.5', 'learning_rate': '0.001', 'epoch': '0.05392', 'train/total_time_seconds': '164.8', 'train/time_per_step_avg': '1.648', 'train/epoch_time_elapsed': '221.1', 'train/estimated_remaining_minutes': '38.45'} +{'eval_loss': '5.443', 'eval_runtime': '16.35', 'eval_samples_per_second': '582.7', 'eval_steps_per_second': '4.587', 'epoch': '0.05392', 'train/total_time_seconds': '164.8', 'train/time_per_step_avg': '1.648', 'train/epoch_time_elapsed': '237.4', 'train/estimated_remaining_minutes': '38.45'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 17.57it/s] + 13%|█████▏ | 200/1500 [07:51<43:21, 2.00s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '21.52', 'grad_norm': '9.062', 'learning_rate': '0.001', 'epoch': '0.06471', 'train/total_time_seconds': '197', 'train/time_per_step_avg': '1.617', 'train/epoch_time_elapsed': '277.5', 'train/estimated_remaining_minutes': '37.76'} +{'loss': '20.92', 'grad_norm': '8.062', 'learning_rate': '0.001', 'epoch': '0.07549', 'train/total_time_seconds': '229.2', 'train/time_per_step_avg': '1.616', 'train/epoch_time_elapsed': '317.3', 'train/estimated_remaining_minutes': '37.11'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '5.031', 'eval_runtime': '16.31', 'eval_samples_per_second': '584.1', 'eval_steps_per_second': '4.598', 'epoch': '0.08088', 'train/total_time_seconds': '245.3', 'train/time_per_step_avg': '1.615', 'train/epoch_time_elapsed': '353.5', 'train/estimated_remaining_minutes': '36.8'} +{'loss': '20.13', 'grad_norm': '3.828', 'learning_rate': '0.001', 'epoch': '0.08628', 'train/total_time_seconds': '261.5', 'train/time_per_step_avg': '1.615', 'train/epoch_time_elapsed': '373.7', 'train/estimated_remaining_minutes': '36.5'} +{'loss': '19.28', 'grad_norm': '5.875', 'learning_rate': '0.001', 'epoch': '0.09706', 'train/total_time_seconds': '293.8', 'train/time_per_step_avg': '1.616', 'train/epoch_time_elapsed': '413.8', 'train/estimated_remaining_minutes': '35.91'} +{'loss': '18.32', 'grad_norm': '3.797', 'learning_rate': '0.001', 'epoch': '0.1078', 'train/total_time_seconds': '326.1', 'train/time_per_step_avg': '1.613', 'train/epoch_time_elapsed': '454', 'train/estimated_remaining_minutes': '35.33'} +{'eval_loss': '4.459', 'eval_runtime': '16.53', 'eval_samples_per_second': '576.4', 'eval_steps_per_second': '4.538', 'epoch': '0.1078', 'train/total_time_seconds': '326.1', 'train/time_per_step_avg': '1.613', 'train/epoch_time_elapsed': '470.6', 'train/estimated_remaining_minutes': '35.33'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 18.81it/s] + 20%|███████▊ | 300/1500 [11:45<40:08, 2.01s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '17.48', 'grad_norm': '3.125', 'learning_rate': '0.001', 'epoch': '0.1186', 'train/total_time_seconds': '358.7', 'train/time_per_step_avg': '1.617', 'train/epoch_time_elapsed': '511.2', 'train/estimated_remaining_minutes': '34.78'} +{'loss': '16.86', 'grad_norm': '3.781', 'learning_rate': '0.001', 'epoch': '0.1294', 'train/total_time_seconds': '391', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '551.3', 'train/estimated_remaining_minutes': '34.22'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '4.038', 'eval_runtime': '16.74', 'eval_samples_per_second': '569', 'eval_steps_per_second': '4.479', 'epoch': '0.1348', 'train/total_time_seconds': '407.2', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '588.1', 'train/estimated_remaining_minutes': '33.93'} +{'loss': '16.16', 'grad_norm': '4.812', 'learning_rate': '0.001', 'epoch': '0.1402', 'train/total_time_seconds': '423.4', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '608.2', 'train/estimated_remaining_minutes': '33.65'} +{'loss': '15.61', 'grad_norm': '2.453', 'learning_rate': '0.001', 'epoch': '0.151', 'train/total_time_seconds': '455.8', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '648.3', 'train/estimated_remaining_minutes': '33.1'} +{'loss': '15.26', 'grad_norm': '5.031', 'learning_rate': '0.001', 'epoch': '0.1618', 'train/total_time_seconds': '488.1', 'train/time_per_step_avg': '1.62', 'train/epoch_time_elapsed': '688.5', 'train/estimated_remaining_minutes': '32.54'} +{'eval_loss': '3.779', 'eval_runtime': '16.48', 'eval_samples_per_second': '578', 'eval_steps_per_second': '4.55', 'epoch': '0.1618', 'train/total_time_seconds': '488.1', 'train/time_per_step_avg': '1.62', 'train/epoch_time_elapsed': '705', 'train/estimated_remaining_minutes': '32.54'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 15.70it/s] + 27%|██████████▍ | 400/1500 [15:39<36:40, 2.00s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '14.86', 'grad_norm': '2.531', 'learning_rate': '0.001', 'epoch': '0.1726', 'train/total_time_seconds': '520.4', 'train/time_per_step_avg': '1.617', 'train/epoch_time_elapsed': '745.3', 'train/estimated_remaining_minutes': '31.98'} +{'loss': '14.42', 'grad_norm': '1.883', 'learning_rate': '0.001', 'epoch': '0.1833', 'train/total_time_seconds': '553', 'train/time_per_step_avg': '1.62', 'train/epoch_time_elapsed': '785.7', 'train/estimated_remaining_minutes': '31.44'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.509', 'eval_runtime': '16.51', 'eval_samples_per_second': '577.1', 'eval_steps_per_second': '4.543', 'epoch': '0.1887', 'train/total_time_seconds': '569.2', 'train/time_per_step_avg': '1.62', 'train/epoch_time_elapsed': '822.2', 'train/estimated_remaining_minutes': '31.17'} +{'loss': '14.03', 'grad_norm': '2.219', 'learning_rate': '0.001', 'epoch': '0.1941', 'train/total_time_seconds': '585.3', 'train/time_per_step_avg': '1.62', 'train/epoch_time_elapsed': '842.4', 'train/estimated_remaining_minutes': '30.89'} +{'loss': '13.66', 'grad_norm': '1.617', 'learning_rate': '0.001', 'epoch': '0.2049', 'train/total_time_seconds': '617.7', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '882.4', 'train/estimated_remaining_minutes': '30.34'} +{'loss': '13.35', 'grad_norm': '1.086', 'learning_rate': '0.001', 'epoch': '0.2157', 'train/total_time_seconds': '650', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '922.6', 'train/estimated_remaining_minutes': '29.79'} +{'eval_loss': '3.323', 'eval_runtime': '16.5', 'eval_samples_per_second': '577.3', 'eval_steps_per_second': '4.545', 'epoch': '0.2157', 'train/total_time_seconds': '650', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '939.1', 'train/estimated_remaining_minutes': '29.79'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 19.13it/s] + 33%|█████████████ | 500/1500 [19:33<33:17, 2.00s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '13.05', 'grad_norm': '2.656', 'learning_rate': '0.001', 'epoch': '0.2265', 'train/total_time_seconds': '682.3', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '979.4', 'train/estimated_remaining_minutes': '29.24'} +{'loss': '12.78', 'grad_norm': '2.047', 'learning_rate': '0.001', 'epoch': '0.2373', 'train/total_time_seconds': '714.9', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '1020', 'train/estimated_remaining_minutes': '28.7'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.121', 'eval_runtime': '16.44', 'eval_samples_per_second': '579.5', 'eval_steps_per_second': '4.562', 'epoch': '0.2427', 'train/total_time_seconds': '731.1', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '1056', 'train/estimated_remaining_minutes': '28.43'} +{'loss': '12.46', 'grad_norm': '2.344', 'learning_rate': '0.001', 'epoch': '0.248', 'train/total_time_seconds': '747.2', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '1076', 'train/estimated_remaining_minutes': '28.16'} +{'loss': '12.18', 'grad_norm': '1.734', 'learning_rate': '0.001', 'epoch': '0.2588', 'train/total_time_seconds': '779.6', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '1116', 'train/estimated_remaining_minutes': '27.61'} +{'loss': '11.92', 'grad_norm': '1.656', 'learning_rate': '0.001', 'epoch': '0.2696', 'train/total_time_seconds': '811.8', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '1156', 'train/estimated_remaining_minutes': '27.06'} +{'eval_loss': '2.951', 'eval_runtime': '16.42', 'eval_samples_per_second': '580.3', 'eval_steps_per_second': '4.568', 'epoch': '0.2696', 'train/total_time_seconds': '811.8', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '1173', 'train/estimated_remaining_minutes': '27.06'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 15.22it/s] +/mnt/data/zainulabideen/zain-exp/notebooks/Activation/exp.py:487: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1831.) + "std": tensor.std().item(), + 40%|███████████████▌ | 600/1500 [23:29<30:04, 2.00s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '11.68', 'grad_norm': '1.523', 'learning_rate': '0.001', 'epoch': '0.2804', 'train/total_time_seconds': '846.4', 'train/time_per_step_avg': '1.641', 'train/epoch_time_elapsed': '1216', 'train/estimated_remaining_minutes': '26.58', 'train/global/act/norm': '2.451e+05', 'train/global/act/mean': '-0.04372', 'train/global/act/std': '1.193', 'train/global/act/max_abs': '47.5', 'train/global/act/frac_near_dtype_limit': '0', 'train/global/act/frac_near_user_limit': '0', 'train/global/grad/norm': '0.8004', 'train/global/grad/mean': '-7.534e-07', 'train/global/grad/std': '0.0001002', 'train/global/grad/max_abs': '0.01471', 'train/global/grad/frac_near_dtype_limit': '0', 'train/global/grad/frac_near_user_limit': '0', 'train/global/param/norm': '202.1', 'train/global/param/mean': '0.001283', 'train/global/param/std': '0.0506', 'train/global/param/max_abs': '1', 'train/global/param/frac_near_dtype_limit': '0', 'train/global/param/frac_near_user_limit': '0', 'train/layer_model_layers_27/act/norm': '2.303e+04', 'train/layer_model_layers_27/act/mean': '0.01982', 'train/layer_model_layers_27/act/std': '1.102', 'train/layer_model_layers_27/act/max_abs': '44.25', 'train/layer_model_layers_27/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_27/act/frac_near_user_limit': '0', 'train/layer_model_layers_27/grad/norm': '0.02829', 'train/layer_model_layers_27/grad/mean': '-3.765e-07', 'train/layer_model_layers_27/grad/std': '3.491e-05', 'train/layer_model_layers_27/grad/max_abs': '0.001175', 'train/layer_model_layers_27/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_27/grad/frac_near_user_limit': '0', 'train/layer_model_layers_87/act/norm': '2.417e+04', 'train/layer_model_layers_87/act/mean': '0.003093', 'train/layer_model_layers_87/act/std': '1.157', 'train/layer_model_layers_87/act/max_abs': '33', 'train/layer_model_layers_87/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_87/act/frac_near_user_limit': '0', 'train/layer_model_layers_87/grad/norm': '0.08349', 'train/layer_model_layers_87/grad/mean': '-6.282e-07', 'train/layer_model_layers_87/grad/std': '0.0001032', 'train/layer_model_layers_87/grad/max_abs': '0.001923', 'train/layer_model_layers_87/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_87/grad/frac_near_user_limit': '0', 'train/layer_model_layers_53/act/norm': '2.161e+04', 'train/layer_model_layers_53/act/mean': '0.0156', 'train/layer_model_layers_53/act/std': '1.036', 'train/layer_model_layers_53/act/max_abs': '43', 'train/layer_model_layers_53/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_53/act/frac_near_user_limit': '0', 'train/layer_model_layers_53/grad/norm': '0.04708', 'train/layer_model_layers_53/grad/mean': '-3.519e-07', 'train/layer_model_layers_53/grad/std': '5.811e-05', 'train/layer_model_layers_53/grad/max_abs': '0.001633', 'train/layer_model_layers_53/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_53/grad/frac_near_user_limit': '0', 'train/layer__model_layers_31/param/norm': '18.88', 'train/layer__model_layers_31/param/mean': '0.001684', 'train/layer__model_layers_31/param/std': '0.04658', 'train/layer__model_layers_31/param/max_abs': '1', 'train/layer__model_layers_31/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_31/param/frac_near_user_limit': '0', 'train/layer_model_layers_24/act/norm': '2.373e+04', 'train/layer_model_layers_24/act/mean': '0.003067', 'train/layer_model_layers_24/act/std': '1.136', 'train/layer_model_layers_24/act/max_abs': '44.5', 'train/layer_model_layers_24/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_24/act/frac_near_user_limit': '0', 'train/layer_model_layers_24/grad/norm': '0.03601', 'train/layer_model_layers_24/grad/mean': '-3.842e-07', 'train/layer_model_layers_24/grad/std': '4.446e-05', 'train/layer_model_layers_24/grad/max_abs': '0.001152', 'train/layer_model_layers_24/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_24/grad/frac_near_user_limit': '0', 'train/layer_model_layers_39/act/norm': '2.249e+04', 'train/layer_model_layers_39/act/mean': '0.01775', 'train/layer_mode +{'loss': '11.46', 'grad_norm': '2.297', 'learning_rate': '0.001', 'epoch': '0.2912', 'train/total_time_seconds': '878.8', 'train/time_per_step_avg': '1.639', 'train/epoch_time_elapsed': '1256', 'train/estimated_remaining_minutes': '26.04'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.814', 'eval_runtime': '16.35', 'eval_samples_per_second': '582.7', 'eval_steps_per_second': '4.587', 'epoch': '0.2966', 'train/total_time_seconds': '894.9', 'train/time_per_step_avg': '1.639', 'train/epoch_time_elapsed': '1292', 'train/estimated_remaining_minutes': '25.76'} +{'loss': '11.25', 'grad_norm': '1.844', 'learning_rate': '0.001', 'epoch': '0.302', 'train/total_time_seconds': '911.1', 'train/time_per_step_avg': '1.639', 'train/epoch_time_elapsed': '1312', 'train/estimated_remaining_minutes': '25.49'} +{'loss': '11.02', 'grad_norm': '1.961', 'learning_rate': '0.001', 'epoch': '0.3128', 'train/total_time_seconds': '943.7', 'train/time_per_step_avg': '1.641', 'train/epoch_time_elapsed': '1352', 'train/estimated_remaining_minutes': '24.95'} +{'loss': '10.83', 'grad_norm': '2.109', 'learning_rate': '0.001', 'epoch': '0.3235', 'train/total_time_seconds': '976', 'train/time_per_step_avg': '1.642', 'train/epoch_time_elapsed': '1393', 'train/estimated_remaining_minutes': '24.4'} +{'eval_loss': '2.694', 'eval_runtime': '16.51', 'eval_samples_per_second': '577', 'eval_steps_per_second': '4.542', 'epoch': '0.3235', 'train/total_time_seconds': '976', 'train/time_per_step_avg': '1.642', 'train/epoch_time_elapsed': '1409', 'train/estimated_remaining_minutes': '24.4'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 17.02it/s] + 47%|██████████████████▏ | 700/1500 [27:23<27:10, 2.04s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '10.67', 'grad_norm': '1.258', 'learning_rate': '0.001', 'epoch': '0.3343', 'train/total_time_seconds': '1008', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '1449', 'train/estimated_remaining_minutes': '23.85'} +{'loss': '10.49', 'grad_norm': '1.938', 'learning_rate': '0.001', 'epoch': '0.3451', 'train/total_time_seconds': '1041', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '1489', 'train/estimated_remaining_minutes': '23.3'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.583', 'eval_runtime': '16.38', 'eval_samples_per_second': '581.5', 'eval_steps_per_second': '4.577', 'epoch': '0.3505', 'train/total_time_seconds': '1057', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '1526', 'train/estimated_remaining_minutes': '23.03'} +{'loss': '10.31', 'grad_norm': '1.477', 'learning_rate': '0.001', 'epoch': '0.3559', 'train/total_time_seconds': '1073', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '1546', 'train/estimated_remaining_minutes': '22.76'} +{'loss': '10.16', 'grad_norm': '1.609', 'learning_rate': '0.001', 'epoch': '0.3667', 'train/total_time_seconds': '1105', 'train/time_per_step_avg': '1.616', 'train/epoch_time_elapsed': '1586', 'train/estimated_remaining_minutes': '22.21'} +{'loss': '10', 'grad_norm': '2.078', 'learning_rate': '0.001', 'epoch': '0.3775', 'train/total_time_seconds': '1138', 'train/time_per_step_avg': '1.616', 'train/epoch_time_elapsed': '1626', 'train/estimated_remaining_minutes': '21.67'} +{'eval_loss': '2.485', 'eval_runtime': '16.64', 'eval_samples_per_second': '572.4', 'eval_steps_per_second': '4.506', 'epoch': '0.3775', 'train/total_time_seconds': '1138', 'train/time_per_step_avg': '1.616', 'train/epoch_time_elapsed': '1643', 'train/estimated_remaining_minutes': '21.67'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 14.85it/s] + 53%|████████████████████▊ | 800/1500 [31:18<23:21, 2.00s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '9.83', 'grad_norm': '1.82', 'learning_rate': '0.001', 'epoch': '0.3882', 'train/total_time_seconds': '1170', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '1684', 'train/estimated_remaining_minutes': '21.13'} +{'loss': '9.726', 'grad_norm': '1.672', 'learning_rate': '0.001', 'epoch': '0.399', 'train/total_time_seconds': '1203', 'train/time_per_step_avg': '1.62', 'train/epoch_time_elapsed': '1724', 'train/estimated_remaining_minutes': '20.58'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.396', 'eval_runtime': '16.48', 'eval_samples_per_second': '578', 'eval_steps_per_second': '4.55', 'epoch': '0.4044', 'train/total_time_seconds': '1219', 'train/time_per_step_avg': '1.62', 'train/epoch_time_elapsed': '1760', 'train/estimated_remaining_minutes': '20.31'} +{'loss': '9.586', 'grad_norm': '1.5', 'learning_rate': '0.001', 'epoch': '0.4098', 'train/total_time_seconds': '1235', 'train/time_per_step_avg': '1.62', 'train/epoch_time_elapsed': '1781', 'train/estimated_remaining_minutes': '20.04'} +{'loss': '9.475', 'grad_norm': '1.492', 'learning_rate': '0.001', 'epoch': '0.4206', 'train/total_time_seconds': '1267', 'train/time_per_step_avg': '1.621', 'train/epoch_time_elapsed': '1821', 'train/estimated_remaining_minutes': '19.5'} +{'loss': '9.304', 'grad_norm': '1.789', 'learning_rate': '0.001', 'epoch': '0.4314', 'train/total_time_seconds': '1300', 'train/time_per_step_avg': '1.62', 'train/epoch_time_elapsed': '1861', 'train/estimated_remaining_minutes': '18.95'} +{'eval_loss': '2.319', 'eval_runtime': '16.36', 'eval_samples_per_second': '582.3', 'eval_steps_per_second': '4.584', 'epoch': '0.4314', 'train/total_time_seconds': '1300', 'train/time_per_step_avg': '1.62', 'train/epoch_time_elapsed': '1877', 'train/estimated_remaining_minutes': '18.95'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 18.96it/s] + 60%|███████████████████████▍ | 900/1500 [35:11<19:54, 1.99s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '9.215', 'grad_norm': '1.969', 'learning_rate': '0.001', 'epoch': '0.4422', 'train/total_time_seconds': '1332', 'train/time_per_step_avg': '1.617', 'train/epoch_time_elapsed': '1918', 'train/estimated_remaining_minutes': '18.41'} +{'loss': '9.11', 'grad_norm': '1.266', 'learning_rate': '0.001', 'epoch': '0.453', 'train/total_time_seconds': '1364', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '1958', 'train/estimated_remaining_minutes': '17.87'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.256', 'eval_runtime': '16.63', 'eval_samples_per_second': '572.7', 'eval_steps_per_second': '4.509', 'epoch': '0.4583', 'train/total_time_seconds': '1381', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '1995', 'train/estimated_remaining_minutes': '17.6'} +{'loss': '8.99', 'grad_norm': '1.531', 'learning_rate': '0.001', 'epoch': '0.4637', 'train/total_time_seconds': '1397', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '2015', 'train/estimated_remaining_minutes': '17.32'} +{'loss': '8.903', 'grad_norm': '1.469', 'learning_rate': '0.001', 'epoch': '0.4745', 'train/total_time_seconds': '1429', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '2055', 'train/estimated_remaining_minutes': '16.78'} +{'loss': '8.798', 'grad_norm': '1.562', 'learning_rate': '0.001', 'epoch': '0.4853', 'train/total_time_seconds': '1461', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '2095', 'train/estimated_remaining_minutes': '16.24'} +{'eval_loss': '2.199', 'eval_runtime': '16.44', 'eval_samples_per_second': '579.6', 'eval_steps_per_second': '4.563', 'epoch': '0.4853', 'train/total_time_seconds': '1461', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '2111', 'train/estimated_remaining_minutes': '16.24'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 15.39it/s] + 67%|█████████████████████████▎ | 1000/1500 [39:05<16:54, 2.03s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '8.712', 'grad_norm': '1.352', 'learning_rate': '0.001', 'epoch': '0.4961', 'train/total_time_seconds': '1494', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '2152', 'train/estimated_remaining_minutes': '15.69'} +{'loss': '8.657', 'grad_norm': '1.375', 'learning_rate': '0.001', 'epoch': '0.5069', 'train/total_time_seconds': '1526', 'train/time_per_step_avg': '1.616', 'train/epoch_time_elapsed': '2192', 'train/estimated_remaining_minutes': '15.15'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.152', 'eval_runtime': '16.52', 'eval_samples_per_second': '576.9', 'eval_steps_per_second': '4.541', 'epoch': '0.5123', 'train/total_time_seconds': '1542', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '2228', 'train/estimated_remaining_minutes': '14.88'} +{'loss': '8.597', 'grad_norm': '1.547', 'learning_rate': '0.001', 'epoch': '0.5177', 'train/total_time_seconds': '1559', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '2248', 'train/estimated_remaining_minutes': '14.61'} +{'loss': '8.528', 'grad_norm': '1.383', 'learning_rate': '0.001', 'epoch': '0.5284', 'train/total_time_seconds': '1591', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '2289', 'train/estimated_remaining_minutes': '14.07'} +{'loss': '8.455', 'grad_norm': '1.391', 'learning_rate': '0.001', 'epoch': '0.5392', 'train/total_time_seconds': '1623', 'train/time_per_step_avg': '1.617', 'train/epoch_time_elapsed': '2329', 'train/estimated_remaining_minutes': '13.53'} +{'eval_loss': '2.105', 'eval_runtime': '16.47', 'eval_samples_per_second': '578.5', 'eval_steps_per_second': '4.554', 'epoch': '0.5392', 'train/total_time_seconds': '1623', 'train/time_per_step_avg': '1.617', 'train/epoch_time_elapsed': '2345', 'train/estimated_remaining_minutes': '13.53'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 17.20it/s] +/mnt/data/zainulabideen/zain-exp/notebooks/Activation/exp.py:487: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1831.) + "std": tensor.std().item(), + 73%|███████████████████████████▊ | 1100/1500 [43:02<13:16, 1.99s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '8.371', 'grad_norm': '1.609', 'learning_rate': '0.001', 'epoch': '0.55', 'train/total_time_seconds': '1658', 'train/time_per_step_avg': '1.639', 'train/epoch_time_elapsed': '2388', 'train/estimated_remaining_minutes': '13', 'train/global/act/norm': '1.817e+05', 'train/global/act/mean': '-0.06364', 'train/global/act/std': '0.882', 'train/global/act/max_abs': '30', 'train/global/act/frac_near_dtype_limit': '0', 'train/global/act/frac_near_user_limit': '0', 'train/global/grad/norm': '0.7324', 'train/global/grad/mean': '-1.664e-07', 'train/global/grad/std': '9.171e-05', 'train/global/grad/max_abs': '0.01361', 'train/global/grad/frac_near_dtype_limit': '0', 'train/global/grad/frac_near_user_limit': '0', 'train/global/param/norm': '225.7', 'train/global/param/mean': '0.001164', 'train/global/param/std': '0.05651', 'train/global/param/max_abs': '1', 'train/global/param/frac_near_dtype_limit': '0', 'train/global/param/frac_near_user_limit': '0', 'train/layer_model_layers_27/act/norm': '1.39e+04', 'train/layer_model_layers_27/act/mean': '0.002256', 'train/layer_model_layers_27/act/std': '0.6657', 'train/layer_model_layers_27/act/max_abs': '17.75', 'train/layer_model_layers_27/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_27/act/frac_near_user_limit': '0', 'train/layer_model_layers_27/grad/norm': '0.0356', 'train/layer_model_layers_27/grad/mean': '-5.762e-07', 'train/layer_model_layers_27/grad/std': '4.393e-05', 'train/layer_model_layers_27/grad/max_abs': '0.00174', 'train/layer_model_layers_27/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_27/grad/frac_near_user_limit': '0', 'train/layer_model_layers_87/act/norm': '1.856e+04', 'train/layer_model_layers_87/act/mean': '-0.00713', 'train/layer_model_layers_87/act/std': '0.8882', 'train/layer_model_layers_87/act/max_abs': '12.12', 'train/layer_model_layers_87/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_87/act/frac_near_user_limit': '0', 'train/layer_model_layers_87/grad/norm': '0.0918', 'train/layer_model_layers_87/grad/mean': '9.888e-07', 'train/layer_model_layers_87/grad/std': '0.0001133', 'train/layer_model_layers_87/grad/max_abs': '0.001129', 'train/layer_model_layers_87/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_87/grad/frac_near_user_limit': '0', 'train/layer_model_layers_53/act/norm': '1.384e+04', 'train/layer_model_layers_53/act/mean': '-0.0009267', 'train/layer_model_layers_53/act/std': '0.6624', 'train/layer_model_layers_53/act/max_abs': '15.31', 'train/layer_model_layers_53/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_53/act/frac_near_user_limit': '0', 'train/layer_model_layers_53/grad/norm': '0.04177', 'train/layer_model_layers_53/grad/mean': '-1.384e-06', 'train/layer_model_layers_53/grad/std': '5.152e-05', 'train/layer_model_layers_53/grad/max_abs': '0.000927', 'train/layer_model_layers_53/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_53/grad/frac_near_user_limit': '0', 'train/layer__model_layers_31/param/norm': '20.27', 'train/layer__model_layers_31/param/mean': '0.001609', 'train/layer__model_layers_31/param/std': '0.05002', 'train/layer__model_layers_31/param/max_abs': '1', 'train/layer__model_layers_31/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_31/param/frac_near_user_limit': '0', 'train/layer_model_layers_24/act/norm': '1.446e+04', 'train/layer_model_layers_24/act/mean': '-0.01199', 'train/layer_model_layers_24/act/std': '0.6924', 'train/layer_model_layers_24/act/max_abs': '17.75', 'train/layer_model_layers_24/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_24/act/frac_near_user_limit': '0', 'train/layer_model_layers_24/grad/norm': '0.0361', 'train/layer_model_layers_24/grad/mean': '-6.869e-07', 'train/layer_model_layers_24/grad/std': '4.46e-05', 'train/layer_model_layers_24/grad/max_abs': '0.001503', 'train/layer_model_layers_24/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_24/grad/frac_near_user_limit': '0', 'train/layer_model_layers_39/act/norm': '1.377e+04', 'train/layer_model_layers_39/act/mean': '0.001056', 'train/layer_m +{'loss': '8.291', 'grad_norm': '1.344', 'learning_rate': '0.001', 'epoch': '0.5608', 'train/total_time_seconds': '1690', 'train/time_per_step_avg': '1.638', 'train/epoch_time_elapsed': '2428', 'train/estimated_remaining_minutes': '12.46'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.071', 'eval_runtime': '16.32', 'eval_samples_per_second': '583.7', 'eval_steps_per_second': '4.595', 'epoch': '0.5662', 'train/total_time_seconds': '1706', 'train/time_per_step_avg': '1.636', 'train/epoch_time_elapsed': '2464', 'train/estimated_remaining_minutes': '12.19'} +{'loss': '8.234', 'grad_norm': '1.461', 'learning_rate': '0.001', 'epoch': '0.5716', 'train/total_time_seconds': '1722', 'train/time_per_step_avg': '1.636', 'train/epoch_time_elapsed': '2484', 'train/estimated_remaining_minutes': '11.91'} +{'loss': '8.194', 'grad_norm': '1.414', 'learning_rate': '0.001', 'epoch': '0.5824', 'train/total_time_seconds': '1755', 'train/time_per_step_avg': '1.64', 'train/epoch_time_elapsed': '2525', 'train/estimated_remaining_minutes': '11.37'} +{'loss': '8.149', 'grad_norm': '1.367', 'learning_rate': '0.001', 'epoch': '0.5932', 'train/total_time_seconds': '1787', 'train/time_per_step_avg': '1.639', 'train/epoch_time_elapsed': '2565', 'train/estimated_remaining_minutes': '10.83'} +{'eval_loss': '2.038', 'eval_runtime': '16.58', 'eval_samples_per_second': '574.5', 'eval_steps_per_second': '4.523', 'epoch': '0.5932', 'train/total_time_seconds': '1787', 'train/time_per_step_avg': '1.639', 'train/epoch_time_elapsed': '2581', 'train/estimated_remaining_minutes': '10.83'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 14.79it/s] + 80%|██████████████████████████████▍ | 1200/1500 [47:30<12:36, 2.52s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '8.088', 'grad_norm': '1.18', 'learning_rate': '0.001', 'epoch': '0.6039', 'train/total_time_seconds': '1819', 'train/time_per_step_avg': '1.618', 'train/epoch_time_elapsed': '2622', 'train/estimated_remaining_minutes': '10.29'} +{'loss': '8.043', 'grad_norm': '1.367', 'learning_rate': '0.001', 'epoch': '0.6147', 'train/total_time_seconds': '1852', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '2662', 'train/estimated_remaining_minutes': '9.746'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.006', 'eval_runtime': '19.6', 'eval_samples_per_second': '486.2', 'eval_steps_per_second': '3.827', 'epoch': '0.6201', 'train/total_time_seconds': '1868', 'train/time_per_step_avg': '1.619', 'train/epoch_time_elapsed': '2702', 'train/estimated_remaining_minutes': '9.475'} +{'loss': '8.001', 'grad_norm': '1.289', 'learning_rate': '0.001', 'epoch': '0.6255', 'train/total_time_seconds': '1890', 'train/time_per_step_avg': '1.675', 'train/epoch_time_elapsed': '2727', 'train/estimated_remaining_minutes': '9.231'} +{'loss': '7.95', 'grad_norm': '1.164', 'learning_rate': '0.001', 'epoch': '0.6363', 'train/total_time_seconds': '1933', 'train/time_per_step_avg': '1.778', 'train/epoch_time_elapsed': '2778', 'train/estimated_remaining_minutes': '8.735'} +{'loss': '7.893', 'grad_norm': '1.211', 'learning_rate': '0.001', 'epoch': '0.6471', 'train/total_time_seconds': '1976', 'train/time_per_step_avg': '1.891', 'train/epoch_time_elapsed': '2830', 'train/estimated_remaining_minutes': '8.234'} +{'eval_loss': '1.98', 'eval_runtime': '20.34', 'eval_samples_per_second': '468.4', 'eval_steps_per_second': '3.687', 'epoch': '0.6471', 'train/total_time_seconds': '1976', 'train/time_per_step_avg': '1.891', 'train/epoch_time_elapsed': '2850', 'train/estimated_remaining_minutes': '8.234'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 16.31it/s] + 87%|████████████████████████████████▉ | 1300/1500 [52:27<08:45, 2.63s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '7.88', 'grad_norm': '1.344', 'learning_rate': '0.001', 'epoch': '0.6579', 'train/total_time_seconds': '2020', 'train/time_per_step_avg': '2.002', 'train/epoch_time_elapsed': '2902', 'train/estimated_remaining_minutes': '7.725'} +{'loss': '7.828', 'grad_norm': '1.273', 'learning_rate': '0.001', 'epoch': '0.6686', 'train/total_time_seconds': '2063', 'train/time_per_step_avg': '2.111', 'train/epoch_time_elapsed': '2954', 'train/estimated_remaining_minutes': '7.209'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '1.956', 'eval_runtime': '20.26', 'eval_samples_per_second': '470.3', 'eval_steps_per_second': '3.702', 'epoch': '0.674', 'train/total_time_seconds': '2084', 'train/time_per_step_avg': '2.163', 'train/epoch_time_elapsed': '2999', 'train/estimated_remaining_minutes': '6.947'} +{'loss': '7.784', 'grad_norm': '1.188', 'learning_rate': '0.001', 'epoch': '0.6794', 'train/total_time_seconds': '2105', 'train/time_per_step_avg': '2.154', 'train/epoch_time_elapsed': '3024', 'train/estimated_remaining_minutes': '6.683'} +{'loss': '7.767', 'grad_norm': '1.203', 'learning_rate': '0.001', 'epoch': '0.6902', 'train/total_time_seconds': '2148', 'train/time_per_step_avg': '2.157', 'train/epoch_time_elapsed': '3075', 'train/estimated_remaining_minutes': '6.154'} +{'loss': '7.714', 'grad_norm': '1.219', 'learning_rate': '0.001', 'epoch': '0.701', 'train/total_time_seconds': '2192', 'train/time_per_step_avg': '2.16', 'train/epoch_time_elapsed': '3127', 'train/estimated_remaining_minutes': '5.621'} +{'eval_loss': '1.935', 'eval_runtime': '20.2', 'eval_samples_per_second': '471.7', 'eval_steps_per_second': '3.713', 'epoch': '0.701', 'train/total_time_seconds': '2192', 'train/time_per_step_avg': '2.16', 'train/epoch_time_elapsed': '3147', 'train/estimated_remaining_minutes': '5.621'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 13.90it/s] + 93%|███████████████████████████████████▍ | 1400/1500 [57:24<04:21, 2.62s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '7.675', 'grad_norm': '1.32', 'learning_rate': '0.001', 'epoch': '0.7118', 'train/total_time_seconds': '2235', 'train/time_per_step_avg': '2.155', 'train/epoch_time_elapsed': '3199', 'train/estimated_remaining_minutes': '5.08'} +{'loss': '7.656', 'grad_norm': '1.211', 'learning_rate': '0.001', 'epoch': '0.7226', 'train/total_time_seconds': '2278', 'train/time_per_step_avg': '2.153', 'train/epoch_time_elapsed': '3250', 'train/estimated_remaining_minutes': '4.534'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '1.912', 'eval_runtime': '19.71', 'eval_samples_per_second': '483.3', 'eval_steps_per_second': '3.804', 'epoch': '0.728', 'train/total_time_seconds': '2301', 'train/time_per_step_avg': '2.164', 'train/epoch_time_elapsed': '3296', 'train/estimated_remaining_minutes': '4.26'} +{'loss': '7.623', 'grad_norm': '1.188', 'learning_rate': '0.001', 'epoch': '0.7334', 'train/total_time_seconds': '2323', 'train/time_per_step_avg': '2.177', 'train/epoch_time_elapsed': '3322', 'train/estimated_remaining_minutes': '3.985'} +{'loss': '7.605', 'grad_norm': '1.109', 'learning_rate': '0.001', 'epoch': '0.7441', 'train/total_time_seconds': '2366', 'train/time_per_step_avg': '2.179', 'train/epoch_time_elapsed': '3373', 'train/estimated_remaining_minutes': '3.429'} +{'loss': '7.559', 'grad_norm': '1.109', 'learning_rate': '0.001', 'epoch': '0.7549', 'train/total_time_seconds': '2410', 'train/time_per_step_avg': '2.175', 'train/epoch_time_elapsed': '3425', 'train/estimated_remaining_minutes': '2.869'} +{'eval_loss': '1.894', 'eval_runtime': '19.41', 'eval_samples_per_second': '490.9', 'eval_steps_per_second': '3.864', 'epoch': '0.7549', 'train/total_time_seconds': '2410', 'train/time_per_step_avg': '2.175', 'train/epoch_time_elapsed': '3444', 'train/estimated_remaining_minutes': '2.869'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 19.84it/s] +100%|████████████████████████████████████| 1500/1500 [1:01:59<00:00, 1.99s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '7.519', 'grad_norm': '1.344', 'learning_rate': '0.001', 'epoch': '0.7657', 'train/total_time_seconds': '2453', 'train/time_per_step_avg': '2.181', 'train/epoch_time_elapsed': '3496', 'train/estimated_remaining_minutes': '2.303'} +{'loss': '7.509', 'grad_norm': '1.367', 'learning_rate': '0.001', 'epoch': '0.7765', 'train/total_time_seconds': '2496', 'train/time_per_step_avg': '2.183', 'train/epoch_time_elapsed': '3547', 'train/estimated_remaining_minutes': '1.734'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '1.876', 'eval_runtime': '19.25', 'eval_samples_per_second': '494.8', 'eval_steps_per_second': '3.895', 'epoch': '0.7819', 'train/total_time_seconds': '2519', 'train/time_per_step_avg': '2.182', 'train/epoch_time_elapsed': '3593', 'train/estimated_remaining_minutes': '1.448'} +{'loss': '7.472', 'grad_norm': '1.078', 'learning_rate': '0.001', 'epoch': '0.7873', 'train/total_time_seconds': '2541', 'train/time_per_step_avg': '2.184', 'train/epoch_time_elapsed': '3619', 'train/estimated_remaining_minutes': '1.16'} +{'loss': '7.451', 'grad_norm': '1.148', 'learning_rate': '0.001', 'epoch': '0.7981', 'train/total_time_seconds': '2576', 'train/time_per_step_avg': '2.101', 'train/epoch_time_elapsed': '3662', 'train/estimated_remaining_minutes': '0.5802'} +{'loss': '7.419', 'grad_norm': '1.117', 'learning_rate': '0.001', 'epoch': '0.8088', 'train/total_time_seconds': '2609', 'train/time_per_step_avg': '1.989', 'train/epoch_time_elapsed': '3702', 'train/estimated_remaining_minutes': '0'} +{'eval_loss': '1.862', 'eval_runtime': '16.4', 'eval_samples_per_second': '581.1', 'eval_steps_per_second': '4.574', 'epoch': '0.8088', 'train/total_time_seconds': '2609', 'train/time_per_step_avg': '1.989', 'train/epoch_time_elapsed': '3718', 'train/estimated_remaining_minutes': '0'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 14.85it/s] +100%|████████████████████████████████████| 1500/1500 [1:01:59<00:00, 2.48s/it] +{'train_runtime': '3719', 'train_samples_per_second': '206.5', 'train_steps_per_second': '0.403', 'train_loss': '11.67', 'epoch': '0.8088', 'train/total_time_seconds': '2609', 'train/time_per_step_avg': '1.989', 'train/epoch_time_elapsed': '3719', 'train/estimated_remaining_minutes': '0'} +100%|██████████████████████████████████████████| 75/75 [00:16<00:00, 4.68it/s] +[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 15.86it/s] +[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 18.91it/s] +Found 7 files to upload + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████░░░░░░░░░░░░ 3 / 7 + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████░░░░░░░░░░░░ 3 / 7 + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████████████████ 7 / 7 ✓ +[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 18.37it/s] +Found 7 files to upload + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████████████████ 7 / 7 ✓ +No files have been modified since last commit. Skipping to prevent empty commit. diff --git a/wandb/run-20260809_035819-cvzjg5ej/files/requirements.txt b/wandb/run-20260809_035819-cvzjg5ej/files/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..123b15ebf857859624f7f4332e92341c8ef13fdf --- /dev/null +++ b/wandb/run-20260809_035819-cvzjg5ej/files/requirements.txt @@ -0,0 +1,149 @@ +asttokens==3.0.1 +comm==0.2.3 +debugpy==1.8.21 +decorator==5.3.1 +executing==2.2.1 +nest-asyncio==1.6.0 +parso==0.8.7 +platformdirs==4.11.0 +psutil==7.2.2 +ptyprocess==0.7.0 +pure_eval==0.2.3 +Pygments==2.20.0 +pyzmq==27.1.0 +setuptools==83.0.0 +six==1.17.0 +tornado==6.5.7 +traitlets==5.15.0 +fsspec==2026.4.0 +wcwidth==0.8.2 +ipython_pygments_lexers==1.1.1 +jedi==0.20.0 +jupyter_core==5.9.1 +matplotlib-inline==0.2.2 +pexpect==4.9.0 +prompt_toolkit==3.0.53 +python-dateutil==2.9.0.post0 +stack_data==0.6.3 +wheel==0.47.0 +jupyter_client==8.9.1 +pip==26.1.2 +ipython==9.15.0 +ipykernel==7.2.0 +threadpoolctl==3.6.0 +pyparsing==3.3.2 +typing_extensions==4.15.0 +Jinja2==3.1.6 +narwhals==2.24.0 +kiwisolver==1.5.0 +joblib==1.5.3 +fonttools==4.63.0 +cycler==0.12.1 +scipy==1.17.1 +pandas==3.0.5 +contourpy==1.3.3 +scikit-learn==1.9.0 +matplotlib==3.11.1 +urllib3==2.7.0 +tqdm==4.70.0 +idna==3.18 +charset-normalizer==3.4.9 +certifi==2026.7.22 +requests==2.34.2 +seaborn==0.13.2 +uv==0.12.0 +shellingham==1.5.4 +mpmath==1.3.0 +attrs==26.1.0 +hf-xet==1.5.2 +nvidia-nccl-cu12==2.21.5 +MarkupSafe==3.0.3 +regex==2026.7.19 +importlib_metadata==9.0.0 +httpcore==1.0.9 +annotated-doc==0.0.5 +multidict==6.7.1 +aiohttp==3.14.3 +aiosignal==1.4.0 +xxhash==3.8.1 +aiohappyeyeballs==2.7.1 +mdurl==0.1.2 +cuda-toolkit==13.0.3.0 +networkx==3.6.1 +PyYAML==6.0.3 +nvidia-cufile==1.15.1.6 +typer==0.27.0 +torchaudio==2.6.0+cu124 +rich==15.0.0 +nvidia-cufft-cu12==11.2.1.3 +h11==0.16.0 +dill==0.4.1 +cuda-pathfinder==1.6.0 +filelock==3.29.0 +nvidia-nvtx-cu12==12.4.127 +httpx==0.28.1 +anyio==4.14.2 +numpy==2.4.4 +yarl==1.24.5 +click==8.4.2 +triton==3.2.0 +frozenlist==1.8.0 +zipp==4.1.0 +propcache==0.5.2 +tokenizers==0.22.2 +markdown-it-py==4.2.0 +nvidia-cuda-runtime==13.0.96 +cuda-bindings==13.3.1 +nvidia-cuda-cupti==13.0.85 +torch==2.6.0+cu124 +multiprocess==0.70.19 +pillow==12.2.0 +transformers==5.15.0.dev0 +wandb==0.28.1 +nvidia-curand==10.4.0.35 +sympy==1.13.1 +nvidia-cusparse==12.6.3.3 +nvidia-cuda-nvrtc==13.0.88 +typing-inspection==0.4.2 +nvidia-cusolver==12.0.4.66 +nvidia-cufft==12.0.0.61 +nvidia-cudnn-cu13==9.20.0.48 +nvidia-cublas==13.1.1.3 +pyarrow==25.0.0 +evaluate==0.4.6 +diffusers==0.39.0 +pydantic==2.13.4 +annotated-types==0.8.0 +protobuf==7.35.1 +sentry-sdk==2.66.1 +einops==0.8.2 +packaging==26.2 +nvidia-nvjitlink-cu12==12.4.127 +nvidia-curand-cu12==10.3.5.147 +nvidia-cusparselt-cu12==0.6.2 +nvidia-cusparse-cu12==12.3.1.170 +nvidia-cuda-runtime-cu12==12.4.127 +torchvision==0.21.0+cu124 +nvidia-cuda-nvrtc-cu12==12.4.127 +nvidia-cuda-cupti-cu12==12.4.127 +nvidia-cusolver-cu12==11.6.1.9 +nvidia-cublas-cu12==12.4.5.8 +nvidia-cudnn-cu12==9.1.0.70 +huggingface_hub==1.26.0 +datasets==5.0.1 +safetensors==0.8.0 +accelerate==1.14.0 +pydantic_core==2.46.4 +ninja==1.13.0 +autocommand==2.2.2 +backports.tarfile==1.2.0 +importlib_metadata==8.7.1 +jaraco.text==4.0.0 +jaraco.context==6.1.0 +jaraco.functools==4.4.0 +more-itertools==10.8.0 +packaging==26.0 +platformdirs==4.4.0 +tomli==2.4.0 +wheel==0.46.3 +zipp==3.23.0 diff --git a/wandb/run-20260809_035819-cvzjg5ej/files/wandb-metadata.json b/wandb/run-20260809_035819-cvzjg5ej/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..3b9eabf2529c6d8ea5e47143e3ce4592c5cb7ace --- /dev/null +++ b/wandb/run-20260809_035819-cvzjg5ej/files/wandb-metadata.json @@ -0,0 +1,96 @@ +{ + "os": "Linux-5.15.0-126-generic-x86_64-with-glibc2.35", + "python": "CPython 3.11.15", + "startedAt": "2026-08-09T03:58:19.280139Z", + "args": [ + "--config", + "configs/baseline.yaml", + "--variants", + "mlp-s10-94L", + "--push" + ], + "program": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py", + "codePath": "sweep.py", + "codePathLocal": "sweep.py", + "git": { + "remote": "https://github.com/deepnevro/Activation.git", + "commit": "34b8d2e8f9a0c5751333310e69fa0c1056381deb" + }, + "email": "deepnevro@gmail.com", + "root": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation", + "host": "deeplens-k3s-node1", + "executable": "/mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python", + "cpu_count": 112, + "cpu_count_logical": 224, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1560765693952", + "used": "708205846528" + } + }, + "memory": { + "total": "2164089937920" + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-39c684a5-fde6-83d7-1663-0859795881ae" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-132944c4-b689-2b5f-89a4-d730401677ab" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bc6c3e3c-9b90-09ca-c034-774961847c54" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea" + } + ], + "cudaVersion": "12.4", + "writerId": "oz75ofyjp8ovu1ftttzp3wyx4l4oquya" +} \ No newline at end of file diff --git a/wandb/run-20260809_035819-cvzjg5ej/files/wandb-summary.json b/wandb/run-20260809_035819-cvzjg5ej/files/wandb-summary.json new file mode 100644 index 0000000000000000000000000000000000000000..4587d24c0b484e5177deacd615ae2291cff30564 --- /dev/null +++ b/wandb/run-20260809_035819-cvzjg5ej/files/wandb-summary.json @@ -0,0 +1 @@ +{"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/norm":4.46875,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/mean":0.0003223419189453125,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/std":0.037353515625,"train/train/tensor_act_model_layers_89_self_attn_k_proj/norm":6000.531589517792,"train/train/layer_model_layers_20/grad/norm":0.02524980048337818,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/norm":0.002251317852734299,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/max_abs":0.000644683837890625,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_down_proj/norm":699.9703697549551,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/norm":0.07727156844774638,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/max_abs":0.00020599365234375,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/max_abs":0.1083984375,"train/train/layer__model_layers_54/param/max_abs":1,"train/train/layer_model_layers_14/grad/std":3.991344609421211e-05,"train/train/tensor_act_model_layers_7_self_attn_q_proj/max_abs":9.75,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_56_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp/norm":632.0394321420083,"train/train/layer__model_layers_71/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/norm":0.0013900842676829366,"train/train/tensor_act_model_layers_69_self_attn/max_abs":1.6171875,"train/train/tensor_act_model_layers_78_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_63/act/max_abs":13.4375,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/norm":0.012581313422166859,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/norm":0.0031682295203920756,"train/train/tensor_act_model_layers_92_self_attn_v_proj/mean":0.00537872314453125,"train/train/tensor_act_model_layers_70_mlp/max_abs":1.453125,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/norm":0.04774744032944506,"train/train/tensor_act_model_layers_64_mlp_up_proj/mean":-0.1029052734375,"train/train/tensor_act_model_layers_10_self_attn_q_proj/max_abs":5.5625,"train/train/tensor_act_model_layers_23_self_attn_v_proj/max_abs":2.359375,"train/train/tensor_act_model_layers_43_mlp_up_proj/max_abs":3.828125,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/norm":0.0368753381487326,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/std":0.0269775390625,"train/train/tensor_act_model_layers_74_mlp_down_proj/mean":0.024200439453125,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_q_proj/max_abs":8.4375,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/std":0.025634765625,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/norm":5.625,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/std":0.00010152785210292422,"train/train/tensor_act_model_layers_82/max_abs":11.875,"train/train/tensor_act_model_layers_5_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp/mean":0.01336669921875,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/norm":6.78125,"train/train/layer_model_layers_91/grad/norm":0.10398106284969418,"train/train/tensor_act_model_layers_80_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_65_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/std":0.033203125,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/max_abs":0.000713348388671875,"train/train/tensor_act_model_layers_35_mlp/std":0.04870621243788334,"train/train/tensor_act_model_layers_73_mlp/norm":955.9582227644396,"train/train/tensor_act_model_layers_4_self_attn_o_proj/max_abs":0.484375,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48/norm":7543.787045821895,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/norm":7.375,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn_q_proj/std":1.056663126648269,"train/train/layer__model_layers_19/param/std":0.04768721983036004,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/max_abs":0.00022792816162109375,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_input_layernorm/max_abs":5.1875,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/norm":0.0018680685851968676,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/mean":-5.439505912363529e-08,"train/train/tensor_act_model_layers_1_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn_v_proj/max_abs":2.515625,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/max_abs":0.12890625,"train/train/tensor_act_model_layers_6_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_78/param/norm":23.43066567022798,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn/mean":-0.0008831024169921875,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp/mean":0.002635955810546875,"train/train/tensor_act_model_layers_56_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_48_mlp/mean":-0.00606536865234375,"train/train/tensor_act_model_layers_61_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/norm":4.375,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/norm":0.023108961239631992,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/std":0.0498046875,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/std":8.023406318439605e-05,"train/train/tensor_act_model_layers_53_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp/norm":3035.3361401297775,"train/train/tensor_act_model_layers_45_self_attn_o_proj/norm":585.174286628416,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/std":0.00012172112112982528,"train/train/tensor_act_model_layers_39_post_attention_layernorm/norm":5792.599365234997,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp_up_proj/mean":-0.088623046875,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_18_post_attention_layernorm/std":1.0000000461004663,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/norm":8.625,"train/train/tensor_act_model_layers_2_self_attn_q_proj/norm":7575.463077088445,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/max_abs":0.0015106201171875,"train/train/tensor_act_model_layers_53_mlp/max_abs":0.8828125,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/max_abs":0.000823974609375,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_input_layernorm/std":0.9990251677228724,"train/train/layer__model_layers_75/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/norm":7.125,"train/train/layer_model_layers_81/act/std":0.8318082435820071,"train/train/tensor_act_model_layers_54_self_attn_k_proj/mean":-0.016510009765625,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/std":5.7251097479006425e-05,"train/train/tensor_act_model_layers_47/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86/mean":0.156982421875,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/norm":3.984375,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_79/grad/mean":1.3425051236115454e-06,"train/train/tensor_act_model_layers_1_self_attn_o_proj/mean":-0.00133514404296875,"train/train/tensor_act_model_layers_88_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/max_abs":0.1279296875,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/std":0.044677734375,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/norm":0.017495810313738456,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/std":7.096113657235705e-05,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/mean":5.650520324707031e-05,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/max_abs":0.00024127960205078125,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/max_abs":0.000392913818359375,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/max_abs":0.1259765625,"train/train/tensor_act_model_layers_34_self_attn_v_proj/norm":2099.5410275296776,"train/train/tensor_grad_model_embed_tokens_weight/std":0.00011924413703655928,"train/train/tensor_act_model_layers_18_self_attn/mean":0.00031387805938720703,"train/train/tensor_act_model_layers_16_self_attn_o_proj/norm":234.52950527272105,"train/train/tensor_act_model_layers_84_self_attn_q_proj/max_abs":6.6875,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/mean":-1.895427703857422e-05,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/norm":0.0015079397446796754,"train/train/tensor_act_model_layers_2_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/norm":0.022684806059533186,"train/train/layer_model_layers_33/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/norm":0.005970458235630448,"train/train/tensor_act_model_layers_1_self_attn/mean":-0.00133514404296875,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/norm":0.02933633054833153,"train/train/layer__model_layers_47/param/norm":21.383770130638798,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/norm":3.09375,"train/train/tensor_act_model_layers_63/std":1.373051150927119,"train/train/tensor_act_model_layers_9_self_attn_k_proj/max_abs":6.40625,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/norm":0.0013873231101420665,"train/train/tensor_act_model_layers_43_mlp/norm":385.98498599280697,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/mean":-0.00052642822265625,"train/train/tensor_act_model_layers_46_input_layernorm/norm":5792.612304688292,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/mean":-0.000484466552734375,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_self_attn_v_proj/norm":1767.0325869770918,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/std":0.04296875,"train/train/layer_model_layers_68/grad/norm":0.07736539156361785,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/max_abs":0.00098419189453125,"train/train/tensor_act_model_layers_93_post_attention_layernorm/max_abs":5.3125,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/norm":0.004289457030294982,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp_down_proj/mean":0.00447845458984375,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/std":9.23726383919341e-05,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/mean":0.00016880035400390625,"train/train/layer__model_layers_1/param/std":0.048227506512630476,"train/train/layer_model_layers_52/act/mean":-0.013035040635329027,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/mean":0.000713348388671875,"train/train/layer__model_layers_47/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_post_attention_layernorm/mean":0.0703125,"train/train/tensor_act_model_layers_81_self_attn/mean":-0.002521514892578125,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/norm":6,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/std":0.04248046875,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/mean":-1.3737007975578308e-06,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/norm":4.09375,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/norm":0.0024995787459349554,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/norm":0.000754534197659066,"train/train/tensor_act_model_layers_81/mean":0.095947265625,"train/train/tensor_act_model_layers_16_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_76_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_k_proj/std":0.740236964730788,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/std":6.253940474246394e-05,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_61/param/max_abs":1,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/max_abs":0.00013065338134765625,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/mean":0.000873565673828125,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/norm":0.004889150321003454,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/norm":5.84375,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/std":6.676670699719835e-05,"train/train/tensor_act_model_layers_64_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_input_layernorm_weight/std":0,"train/train/layer__model_layers_57/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_22/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_q_proj/std":1.439457100147157,"train/train/tensor_param_model_layers_0_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/max_abs":0.00010776519775390625,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp_down_proj/max_abs":0.85546875,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/std":0.05517578125,"train/train/layer_model_layers_5/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_input_layernorm/std":1.0000000949948982,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/std":0.04052734375,"train/train/layer_model_layers_20/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/norm":0.0008132465771302361,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp/mean":-0.0007801055908203125,"train/train/tensor_act_model_layers_64_self_attn_k_proj/max_abs":4.78125,"train/train/tensor_act_model_layers_62_post_attention_layernorm/norm":5792.606445315906,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/std":4.107537180054505e-05,"train/train/tensor_param_model_layers_16_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_62_mlp_down_proj/norm":658.949244658078,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_32/grad/mean":-5.535253428224933e-07,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/std":0.033447265625,"train/train/tensor_param_model_layers_11_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/max_abs":0.00040435791015625,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/norm":0.004186723660272024,"train/train/layer__model_layers_66/param/norm":23.019480540946184,"train/train/tensor_act_model_layers_22_self_attn/max_abs":0.88671875,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/std":3.256961147711553e-05,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/norm":3.953125,"train/train/tensor_act_model_layers_12_post_attention_layernorm/std":1.0000000510676714,"train/train/tensor_act_model_layers_28_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/std":0.00015195600155700265,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/mean":7.890164852142334e-06,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/mean":1.1481461115181446e-08,"train/train/tensor_act_model_layers_37_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/norm":0.030331525141318774,"train/train/tensor_act_model_layers_19/max_abs":17.875,"train/train/tensor_param_model_layers_87_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/max_abs":0.00160980224609375,"train/train/tensor_act_model_layers_83_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_v_proj/std":0.4179687733962151,"train/train/tensor_act_model_layers_73_self_attn/std":0.0529795311946369,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/std":6.69420052172679e-05,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/norm":5.90625,"train/train/tensor_act_model_layers_23_self_attn_k_proj/norm":5270.495363830182,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_76/param/frac_near_user_limit":0,"train/train/layer__model_layers_91/param/frac_near_user_limit":0,"train/train/layer__model_layers_4/param/norm":19.253436685920516,"train/train/tensor_act_model_layers_51_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/std":0.0439453125,"train/train/tensor_act_model_layers_13_mlp_down_proj/max_abs":0.5390625,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp/max_abs":2.046875,"train/train/tensor_act_model_layers_39_self_attn_k_proj/norm":4793.477635924666,"train/train/tensor_act_model_layers_22_mlp/norm":256.49137022046824,"train/train/tensor_act_model_layers_73_mlp_down_proj/max_abs":2.0625,"train/train/tensor_act_model_layers_21_mlp_up_proj/norm":2408.703144305664,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/mean":1.334119588136673e-07,"train/train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/mean":2.3366883397102356e-06,"train/train/layer_model_layers_80/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp_down_proj/max_abs":17.875,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/std":8.720754334560032e-05,"train/train/tensor_param_model_layers_55_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/std":0.033447265625,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_3_self_attn_o_proj/std":0.12451174703298509,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/norm":5,"train/train/layer_model_layers_10/act/std":0.7248507592998292,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/max_abs":0.0002994537353515625,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_v_proj/mean":0.00518798828125,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn/mean":-0.0002884864807128906,"train/train/tensor_act_model_layers_1_self_attn_k_proj/mean":0.03070068359375,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn_k_proj/mean":0.0821533203125,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/max_abs":0.0002613067626953125,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_76/grad/mean":2.659348750635168e-06,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/max_abs":0.0005340576171875,"train/train/tensor_act_model_layers_90_self_attn_o_proj/max_abs":2.59375,"train/train/tensor_act_model_layers_44_self_attn/mean":-0.0004942417144775391,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/std":0.046142578125,"train/train/tensor_act_model_layers_52_self_attn_k_proj/max_abs":5.1875,"train/train/tensor_act_model_layers_10_self_attn_o_proj/max_abs":0.8125,"train/train/tensor_act_model_layers_30_self_attn_v_proj/std":0.34960939470345526,"train/train/tensor_act_model_layers_65_self_attn_q_proj/std":1.0136801037380896,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/norm":0.028538684143849902,"train/train/tensor_act_model_layers_20_self_attn_q_proj/mean":0.02886962890625,"train/train/layer_model_layers_0/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/std":0.03466796875,"train/train/layer_model_layers_62/grad/std":7.260981439795395e-05,"train/train/estimated_remaining_minutes":0,"train/train/layer__model_layers_60/param/std":0.05560692608651663,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_post_attention_layernorm/mean":0.04290771484375,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/norm":5.03125,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/std":0.0234375,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/std":4.910371983756954e-05,"train/train/layer_model_layers_7/act/mean":-0.0011301040649414062,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/mean":1.0829535312950611e-07,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/std":0.0361328125,"train/train/tensor_act_model_layers_33_self_attn_k_proj/norm":4694.834419135524,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_66/act/mean":-0.02051118703988882,"train/train/tensor_act_model_layers_5_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/mean":1.99128407984972e-07,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/max_abs":0.0002193450927734375,"train/train/tensor_act_model_layers_33/max_abs":17.625,"train/train/tensor_act_model_layers_14_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_mlp/std":0.040527346622512894,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/mean":3.337860107421875e-05,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_input_layernorm/mean":0.05816650390625,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_10/act/norm":15144.287275254206,"train/train/layer_model_layers_80/grad/std":9.788855905227101e-05,"train/train/layer_model_layers_61/act/mean":-0.0014203878549429087,"train/train/tensor_act_model_layers_69_mlp/norm":972.8257132627101,"train/train/tensor_act_model_layers_56_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/max_abs":0.2021484375,"train/train/tensor_act_model_layers_75_self_attn_k_proj/max_abs":5.8125,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/std":7.212808922801445e-05,"train/train/tensor_act_model_layers_2_self_attn_v_proj/std":0.2714843801457247,"train/train/tensor_act_model_layers_83_self_attn/mean":0.001979827880859375,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82/std":1.7636794241134268,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_mlp/max_abs":16.375,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/mean":0.012359619140625,"train/train/tensor_act_model_layers_75_self_attn_o_proj/mean":-0.00296783447265625,"train/train/tensor_param_model_layers_41_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_8_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_55/param/norm":21.745464685193554,"train/train/tensor_grad_model_norm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/std":0.037109375,"train/train/tensor_act_model_layers_63_input_layernorm/max_abs":5.875,"train/train/tensor_act_model_layers_3_self_attn_k_proj/norm":8525.439366498866,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/max_abs":0.00019168853759765625,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/norm":0.008339678563932145,"train/train/layer__model_layers_70/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/std":9.46921904674118e-05,"train/train/layer__model_layers_51/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/std":7.974363406764388e-06,"train/train/tensor_act_model_layers_63_mlp/max_abs":1.03125,"train/train/tensor_act_model_layers_61_mlp_up_proj/std":0.43359416239950826,"train/train/tensor_act_model_layers_82_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/max_abs":0.0001010894775390625,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_38/grad/std":5.9017827534993704e-05,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_o_proj/max_abs":1.765625,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/max_abs":0.001739501953125,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/mean":-1.993030309677124e-07,"train/train/tensor_act_model_layers_10_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/std":9.060093326682747e-05,"train/train/tensor_act_model_layers_44_self_attn_q_proj/std":0.9462911174740046,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp_up_proj/norm":5649.40424468466,"train/train/layer_model_layers_11/grad/std":4.007609189918438e-05,"train/train/tensor_act_model_layers_11_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82/norm":10246.656737941421,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/std":8.255527940351763e-05,"train/train/layer_model_layers_13/grad/std":4.0257462876667426e-05,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/norm":0.009446804913163975,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_8/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn/norm":909.4365599799609,"train/train/tensor_act_model_layers_40_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/std":9.935386018213107e-05,"train/train/tensor_act_model_layers_4_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp_up_proj/max_abs":4.21875,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/std":0.033935546875,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/max_abs":0.000514984130859375,"train/train/layer_model_layers_34/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/norm":7.90625,"train/train/tensor_act_model_layers_64_input_layernorm/mean":0.0552978515625,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_80/act/mean":0.00289612550001878,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_post_attention_layernorm/max_abs":6.09375,"train/train/layer__model_layers_65/param/mean":0.001354163968804846,"train/train/tensor_act_model_layers_21_self_attn_k_proj/mean":-0.0333251953125,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/mean":-8.177012205123901e-07,"train/train/tensor_act_model_layers_30_mlp_up_proj/mean":-0.0673828125,"train/train/tensor_act_model_layers_80_input_layernorm/max_abs":5.71875,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/std":5.774332965964883e-05,"train/train/tensor_act_model_layers_25_self_attn_o_proj/std":0.05763428783327777,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/norm":0.0008305906543690597,"train/train/tensor_act_model_layers_43_mlp_down_proj/std":0.06652866336642194,"train/train/tensor_act_model_layers_29_mlp_down_proj/mean":0.01116943359375,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/std":2.770608243018693e-05,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/std":0.0247802734375,"train/train/tensor_act_model_layers_78/frac_near_user_limit":0,"train/train/layer_model_layers_1/grad/max_abs":0.002960205078125,"train/train/layer_model_layers_35/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/mean":-6.274785846471786e-08,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_input_layernorm/std":1.0000023003641136,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_77/grad/max_abs":0.00171661376953125,"train/train/tensor_act_model_layers_18_self_attn_v_proj/norm":1631.3799332827969,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn_q_proj/norm":5541.272492840632,"train/train/tensor_act_model_layers_39/norm":7559.0359862208825,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/std":9.793355433608928e-05,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/max_abs":0.2578125,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/max_abs":0.0003833770751953125,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/norm":0.023286011996108788,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/norm":5.28125,"train/train/tensor_act_model_layers_69_self_attn/std":0.13605561299317123,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/max_abs":0.0002460479736328125,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_0/norm":8470.84941793658,"train/train/layer__model_layers_47/param/max_abs":1,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/std":4.871001241503894e-05,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_input_layernorm/std":0.9970739814772993,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/mean":-0.000118255615234375,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/max_abs":0.0003509521484375,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/norm":0.0020150125029435723,"train/train/tensor_act_model_layers_35_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_input_layernorm/max_abs":6.0625,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/mean":1.2993812561035156e-05,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/std":0.026123046875,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/max_abs":0.00147247314453125,"train/train/tensor_act_model_layers_28/std":1.3203241189750143,"train/train/layer_model_layers_71/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_k_proj/mean":0.02197265625,"train/train/layer__model_layers_52/param/mean":0.0013176833225672776,"train/train/tensor_act_model_layers_63_mlp/norm":661.8512047819681,"train/train/tensor_act_model_layers_47_self_attn_o_proj/norm":992.1523644585675,"train/train/tensor_act_model_layers_87_mlp/norm":1820.7761108830575,"train/train/tensor_act_model_layers_56_self_attn_k_proj/max_abs":5.65625,"train/train/layer_model_layers_46/act/norm":14415.54972815357,"train/train/tensor_act_model_layers_42_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/norm":6.5,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_4/grad/std":5.002366157998068e-05,"train/train/tensor_act_model_layers_21/mean":0.03704833984375,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/norm":5.4375,"train/train/tensor_act_model_layers_73_self_attn_o_proj/norm":307.2963042317912,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_13_self_attn/std":0.03698751127455635,"train/train/tensor_act_model_layers_84_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/std":3.927582066796292e-05,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_k_proj/norm":4557.180364211467,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/mean":8.335337042808533e-08,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/std":0.05126953125,"train/train/tensor_act_model_layers_52_mlp_up_proj/mean":-0.09033203125,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/norm":0.00039420909707721277,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp_down_proj/mean":-0.0084686279296875,"train/train/tensor_act_model_layers_58_self_attn_k_proj/mean":0.015350341796875,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_mlp_up_proj/max_abs":3.5625,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/mean":-2.6673078536987305e-06,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/std":2.3663395158219575e-05,"train/train/tensor_act_model_layers_92_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/mean":0.0003681182861328125,"train/train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/norm":0.00234614736414447,"train/train/tensor_act_model_layers_9_self_attn_v_proj/norm":2187.9060135436,"train/train/tensor_act_model_layers_57_mlp_down_proj/mean":-0.003173828125,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/std":0.00016833901598302308,"train/train/tensor_act_model_layers_4_mlp_down_proj/norm":387.6566382657851,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/norm":4.75,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/norm":0.01828825547410712,"train/train/tensor_act_model_layers_17_self_attn/mean":-0.0008487701416015625,"train/train/tensor_act_model_layers_57_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp/std":0.23144531853591332,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/mean":2.891756594181061e-07,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_0/param/mean":0.001378105508741834,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_input_layernorm/mean":0.004070281982421875,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn/std":0.23952149475257986,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/std":0.052001953125,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/max_abs":0.0001239776611328125,"train/train/layer__model_layers_86/param/norm":23.721426067745167,"train/train/layer__model_layers_8/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp/norm":390.37375003261207,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn_k_proj/max_abs":5.96875,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/std":2.2258773819564516e-05,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/mean":-0.000507354736328125,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/max_abs":0.0010986328125,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/std":0.0311279296875,"train/train/tensor_param_model_layers_9_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_29/grad/std":4.543201059753852e-05,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/std":0.00010969824787823749,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_o_proj/max_abs":2.40625,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/mean":2.1257437765598297e-07,"train/train/tensor_act_model_layers_74_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/std":0.000133286523646526,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/mean":0.0002841949462890625,"train/train/layer_model_layers_9/act/mean":-0.018515568513136644,"train/train/tensor_param_model_layers_85_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/norm":11.3125,"_step":106,"train/train/tensor_act_model_layers_25_post_attention_layernorm/std":1.0000000055879352,"train/train/tensor_act_model_layers_20_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_69/grad/norm":0.0643847984551902,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/mean":-3.8623809814453125e-05,"train/train/tensor_act_model_layers_11_self_attn_k_proj/norm":5887.045601506922,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_4/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/max_abs":0.1455078125,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/norm":0.01837371104124628,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_37/grad/mean":-5.115656103712162e-07,"train/train/tensor_act_model_layers_10/max_abs":18.375,"train/train/tensor_act_model_layers_60_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_83_post_attention_layernorm/std":0.9970739814772993,"train/train/tensor_act_model_layers_50/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/norm":0.00036321221893937576,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_53/norm":7591.393989703343,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/norm":0.0013535188833833259,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_input_layernorm/std":0.9960939519545406,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/norm":7.375,"train/train/tensor_act_model_layers_35_self_attn_v_proj/mean":-0.0007076263427734375,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/max_abs":0.00035858154296875,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/mean":-1.6426201909780502e-07,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_90_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/norm":6.1875,"train/train/tensor_act_model_layers_71_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_69/act/norm":15072.794161335558,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/std":0.048095703125,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/max_abs":0.000637054443359375,"train/train/tensor_act_model_layers_57_mlp/std":0.09887726218563803,"train/train/tensor_act_model_layers_14_self_attn_k_proj/std":0.9775409069672675,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/max_abs":0.2177734375,"train/train/tensor_act_model_layers_81_self_attn_v_proj/std":0.4794936484903247,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/std":5.144553267358216e-05,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_k_proj/norm":6234.327043722333,"train/train/layer_model_layers_17/grad/std":4.545912450728324e-05,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/norm":5.40625,"train/train/layer_model_layers_86/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/max_abs":0.2294921875,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/max_abs":0.28515625,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/max_abs":0.0003814697265625,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/mean":-2.473592758178711e-05,"train/train/tensor_act_model_layers_77_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn_v_proj/std":0.4902343832520374,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/mean":-7.915496826171875e-05,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_k_proj/norm":5421.075648739972,"train/train/tensor_act_model_layers_91_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_q_proj/norm":5883.241931749809,"train/train/tensor_param_model_layers_22_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_77/std":1.599616551412244,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/max_abs":0.173828125,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn/max_abs":0.671875,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/norm":0.005170568504289986,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/std":6.750792045733886e-05,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77/max_abs":12.5,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/norm":0.04354455256898999,"train/train/tensor_act_model_layers_27_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/std":0.00014021667315512656,"train/train/tensor_act_model_layers_73_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/max_abs":0.1748046875,"train/train/tensor_act_model_layers_59_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_7/grad/std":6.78591416744602e-05,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/std":0.04638671875,"train/train/layer_model_layers_60/grad/norm":0.05527187141974303,"train/train/tensor_act_model_layers_45_mlp_down_proj/max_abs":0.82421875,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/std":7.800925770717344e-05,"train/train/tensor_act_model_layers_58_mlp/max_abs":1.3984375,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/std":0.033935546875,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/norm":0.01380454137494868,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/std":0.047119140625,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn_v_proj/norm":2795.5397065945326,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer__model_layers_28/param/std":0.050096250035000775,"train/train/tensor_act_model_layers_69_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/max_abs":0.00023937225341796875,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/mean":-3.331899642944336e-05,"train/train/tensor_act_model_layers_78_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn_v_proj/max_abs":1.765625,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/max_abs":0.0004520416259765625,"train/train/tensor_act_model_layers_74_self_attn_o_proj/norm":618.9524845050278,"train/train/tensor_act_model_layers_47_self_attn_o_proj/max_abs":1.4609375,"train/train/tensor_act_model_layers_73_mlp_up_proj/mean":-0.077392578125,"train/train/tensor_act_model_layers_20/norm":7890.851924064829,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/max_abs":0.1630859375,"train/train/tensor_param_model_layers_80_input_layernorm_weight/std":0,"train/train/layer__model_layers_63/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp_up_proj/max_abs":3.65625,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/std":2.4360474889725685e-05,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn_k_proj/max_abs":5.125,"train/train/tensor_act_model_layers_24_self_attn_k_proj/mean":0.0011057853698730469,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/mean":-0.00017547607421875,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/max_abs":0.255859375,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/norm":7.0625,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/norm":0.018984386616770012,"train/train/tensor_act_model_layers_40_self_attn_o_proj/std":0.09253101271404147,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_49_mlp_up_proj/mean":-0.093505859375,"train/train/tensor_act_model_layers_60_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/norm":5.15625,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/std":0.038330078125,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/mean":-1.069623976945877e-06,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/norm":6.28125,"train/train/layer_model_layers_32/act/mean":-0.0017975981418903058,"train/train/tensor_act_model_layers_73_input_layernorm/mean":0.036376953125,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/mean":-3.457069396972656e-05,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/max_abs":0.00017642974853515625,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp_down_proj/max_abs":2.28125,"train/train/tensor_act_model_layers_66_mlp_up_proj/norm":4821.758714361429,"train/train/tensor_act_model_layers_18_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/std":7.518730069758797e-05,"train/train/tensor_act_model_layers_26_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp_up_proj/mean":-0.0994873046875,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/max_abs":0.09912109375,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/std":0.05029296875,"train/train/tensor_act_model_layers_46_post_attention_layernorm/std":0.9960938322777808,"train/train/tensor_act_model_layers_5/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/mean":0.0001659393310546875,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/mean":0.0003719329833984375,"train/train/tensor_act_model_layers_80_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_0/param/max_abs":1,"train/train/layer_model_layers_44/grad/max_abs":0.002105712890625,"train/train/layer_model_layers_52/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_o_proj/norm":410.7656186242378,"train/train/tensor_act_model_layers_13_self_attn/norm":214.72004332620855,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/max_abs":0.228515625,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_k_proj/norm":4462.538391516322,"train/train/tensor_act_model_layers_18_self_attn_o_proj/mean":0.00031387805938720703,"train/train/tensor_act_model_layers_74_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/norm":5.5625,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_16/grad/norm":0.02985986391429142,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/norm":0.024459719880707647,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/std":1.7097584153052386e-05,"train/train/tensor_act_model_layers_71_self_attn_k_proj/norm":4351.057300814402,"train/train/layer__model_layers_18/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/norm":0.0684835124256314,"train/train/layer_model_layers_35/grad/std":5.001362093369733e-05,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/mean":-3.960516892220767e-08,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/max_abs":0.00051116943359375,"train/train/tensor_act_model_layers_29_self_attn/max_abs":0.59375,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_o_proj/norm":690.4142818059483,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_49/param/max_abs":1,"train/train/tensor_act_model_layers_57_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_13/grad/max_abs":0.0013885498046875,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/std":5.2875835902784576e-05,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_post_attention_layernorm/mean":0.04876708984375,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/std":3.48364913657742e-05,"train/train/tensor_act_model_layers_3_mlp/norm":497.63441981347637,"train/train/layer__model_layers_13/param/std":0.04860898512846425,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_k_proj/mean":0.05322265625,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/max_abs":0.0003299713134765625,"train/train/tensor_act_model_layers_93_mlp_down_proj/max_abs":16.375,"train/train/tensor_act_model_layers_78_self_attn_k_proj/mean":-0.02685546875,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17/max_abs":18,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_42/grad/mean":-5.988636622190848e-07,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89/norm":12606.283880890294,"train/train/tensor_act_model_layers_84_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_87_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_mlp_down_proj/norm":385.98498599280697,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_44/act/std":0.675780516560674,"train/train/tensor_act_model_layers_18_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp/max_abs":0.8671875,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/mean":1.6728881746530533e-07,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/norm":3.546875,"train/train/tensor_act_model_layers_14_self_attn_q_proj/mean":-0.00665283203125,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/std":0.0224609375,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/mean":1.2665987014770508e-07,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/std":0.03466796875,"train/train/tensor_act_model_layers_70_post_attention_layernorm/max_abs":5.96875,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/std":0.041259765625,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/max_abs":0.14453125,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/norm":0.0067498025666608875,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/mean":8.131610229611397e-08,"train/train/tensor_act_model_layers_80_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/norm":5.25,"train/train/layer_model_layers_89/act/norm":19773.736440139248,"train/train/tensor_act_model_layers_81_self_attn_q_proj/max_abs":6.28125,"train/train/tensor_act_model_layers_19_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/mean":-4.507601261138916e-06,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/max_abs":0.00023651123046875,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/max_abs":0.0002765655517578125,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_50/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/norm":6.34375,"train/train/layer__model_layers_69/param/std":0.05596413925610011,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_47/param/mean":0.0013968695344493468,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/mean":-2.1886080503463745e-06,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/max_abs":0.00023174285888671875,"train/train/tensor_act_model_layers_7/norm":8515.525395762596,"train/train/tensor_act_model_layers_8_self_attn/std":0.07007054894830737,"train/train/tensor_act_model_layers_27_self_attn_k_proj/mean":0.0458984375,"train/train/tensor_act_model_layers_11_mlp/norm":236.8761024315632,"train/train/tensor_act_model_layers_18_self_attn_q_proj/std":0.9785182061989369,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_35_self_attn/std":0.12427184640003192,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/std":3.643256671207183e-05,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_30/grad/norm":0.03953976450365753,"train/train/tensor_act_model_layers_87_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/norm":0.009753076120050592,"train/train/tensor_act_model_layers_61/mean":0.0762939453125,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/norm":0.0015910809170554675,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/norm":9.75,"train/train/tensor_act_model_layers_44_mlp_down_proj/mean":0.0088348388671875,"train/train/layer__model_layers_63/param/norm":22.665343002864528,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/norm":0.0024772515985353517,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/norm":4.90625,"train/train/tensor_act_model_layers_67_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/norm":3.78125,"train/train/tensor_act_model_layers_54_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_88/act/mean":-0.01352223983177772,"train/train/layer__model_layers_41/param/mean":0.0015948410153203002,"train/train/layer_model_layers_72/act/norm":14891.432296044368,"train/train/tensor_act_model_layers_25_mlp_up_proj/norm":2600.4597268032326,"train/train/layer_model_layers_75/grad/mean":2.363933520551405e-06,"train/train/tensor_act_model_layers_37_mlp_up_proj/norm":3046.748384255412,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/mean":3.015156835317612e-08,"train/train/tensor_act_model_layers_63_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/max_abs":0.1611328125,"train/train/tensor_act_model_layers_34_self_attn_o_proj/mean":-0.0007953643798828125,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/std":3.8428926419788436e-05,"train/train/tensor_act_model_layers_88_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/norm":5.25,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_post_attention_layernorm/mean":0.0361328125,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/norm":3.046875,"train/train/tensor_act_model_layers_91_self_attn_q_proj/mean":0.020233154296875,"train/train/tensor_act_model_layers_52/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/max_abs":0.25390625,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/max_abs":0.00020313262939453125,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/max_abs":0.0004253387451171875,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/std":0.00016761569020778638,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/max_abs":0.212890625,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/mean":6.474554538726807e-06,"train/train/tensor_act_model_layers_17_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_q_proj/std":1.0703128132506419,"train/train/tensor_act_model_layers_10_self_attn_o_proj/norm":392.23420021884203,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/std":2.024446066023912e-05,"train/train/tensor_act_model_layers_3_self_attn_q_proj/norm":9358.961556894226,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/norm":5.25,"train/train/tensor_param_model_layers_85_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/max_abs":0.00060272216796875,"train/train/tensor_act_model_layers_47_mlp_down_proj/mean":-0.00173187255859375,"train/train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn_o_proj/mean":-0.000195235013961792,"train/train/tensor_act_model_layers_69_input_layernorm/mean":0.03875732421875,"train/train/tensor_act_model_layers_52_self_attn_q_proj/std":0.7880878631634448,"train/train/tensor_act_model_layers_44_self_attn_o_proj/norm":533.673284835028,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/max_abs":0.0004863739013671875,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/std":0.0250244140625,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/mean":9.610084816813469e-08,"train/train/layer_model_layers_62/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/std":6.954033316222978e-05,"train/train/tensor_act_model_layers_62_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn/mean":-0.003444671630859375,"train/train/tensor_act_model_layers_22_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/norm":0.01112561411448462,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/norm":0.033218604574665696,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/std":2.970515439745493e-05,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/mean":-1.5478581190109253e-06,"train/train/tensor_act_model_layers_46_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/std":2.8955667597033176e-05,"train/train/tensor_act_model_layers_15_self_attn_k_proj/mean":-0.0220947265625,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/mean":-1.0826624929904938e-08,"train/train/tensor_act_model_layers_35_mlp/mean":0.006988525390625,"train/train/tensor_act_model_layers_72_self_attn_v_proj/std":0.40039093896348166,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/std":0.0220947265625,"train/train/tensor_act_model_layers_20_input_layernorm/mean":0.028656005859375,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/mean":-2.6426278054714203e-08,"train/train/tensor_act_model_layers_53_post_attention_layernorm/mean":0.04547119140625,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/max_abs":0.000186920166015625,"train/train/layer__model_layers_24/param/mean":0.0016106235069715288,"train/train/tensor_act_model_layers_14_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_v_proj/std":0.4516611758320304,"train/train/tensor_act_model_layers_89/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/mean":0.000446319580078125,"train/train/tensor_act_model_layers_56_self_attn_o_proj/std":0.05646045046525053,"train/train/tensor_act_model_layers_3_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_q_proj/max_abs":5.46875,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/max_abs":0.126953125,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/max_abs":0.000423431396484375,"train/train/tensor_act_model_layers_29_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21/max_abs":17.875,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_input_layernorm/mean":0.03253173828125,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/std":0.06396484375,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/norm":0.00041669020149358917,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81/max_abs":11.9375,"train/train/tensor_act_model_layers_11_self_attn/std":0.04339730185057409,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/max_abs":0.002960205078125,"train/train/tensor_param_model_layers_48_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_72/std":1.5000131961619543,"train/train/tensor_act_model_layers_74_post_attention_layernorm/std":1.0000018682313396,"train/train/tensor_act_model_layers_44_mlp_down_proj/max_abs":0.91796875,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/norm":2.75,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/std":0.029052734375,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/norm":0.013901652904987022,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/max_abs":0.00103759765625,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/norm":0.0005325056413961861,"train/train/layer__model_layers_2/param/mean":0.0013782899948810452,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/mean":0.00011205673217773438,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn_q_proj/norm":5697.203980936033,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/std":2.854151177188653e-05,"train/train/tensor_act_model_layers_26_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_67/act/max_abs":13.3125,"train/train/tensor_act_model_layers_21_self_attn_k_proj/max_abs":4.28125,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/norm":5.625,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/max_abs":0.1904296875,"train/train/tensor_act_model_layers_67/std":1.4472813624711713,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/std":0.032470703125,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_lm_head/mean":-2.2265625,"train/train/tensor_act_model_layers_4_self_attn/mean":0.0008726119995117188,"train/train/layer_model_layers_29/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28/norm":7658.182113449786,"train/train/layer__model_layers_11/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/norm":6.90625,"train/train/tensor_act_model_layers_1_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/max_abs":0.205078125,"train/train/tensor_act_model_layers_43_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/max_abs":0.22265625,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/std":0.032470703125,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_36/param/std":0.05112054314294771,"train/train/tensor_act_model_layers_23_mlp_up_proj/norm":2410.443822442326,"train/train/tensor_act_model_layers_47_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/max_abs":0.1494140625,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/std":0.00015428319361259174,"train/train/tensor_act_model_layers_49_mlp_down_proj/norm":465.9405843886969,"train/train/tensor_act_model_layers_11_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp_up_proj/std":0.5117190266382765,"train/train/tensor_act_model_layers_31_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/std":0.031982421875,"train/train/tensor_act_model_layers_24_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/mean":1.2032687664031982e-06,"train/train/tensor_act_model_layers_30_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/max_abs":0.00052642822265625,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/max_abs":0.0017242431640625,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/mean":-0.0004367828369140625,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/std":2.207831023721957e-05,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/std":1.2490340504355464e-05,"train/train/tensor_act_model_layers_81_self_attn_o_proj/norm":1258.459225299919,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_input_layernorm/norm":5792.614624035048,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/norm":0.002962318138156912,"train/train/tensor_act_model_layers_72_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_input_layernorm/norm":5792.607299809795,"train/train/layer__model_layers_1/param/mean":0.001432326580172582,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/mean":4.280358552932739e-06,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/max_abs":0.14453125,"train/train/tensor_act_model_layers_81_self_attn_q_proj/norm":6817.246547052309,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/norm":0.03076716783490094,"train/train/tensor_act_model_layers_84_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/max_abs":0.00201416015625,"train/train/tensor_act_model_layers_45_mlp/std":0.07971219672767066,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/max_abs":0.107421875,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/norm":7.375,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp_down_proj/mean":0.0005970001220703125,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/mean":1.2470409274101257e-06,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/mean":1.6030389815568924e-06,"train/train/tensor_act_model_layers_55_self_attn/max_abs":1.6015625,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/max_abs":0.2431640625,"train/train/tensor_act_model_layers_48_self_attn_k_proj/std":0.9062501709779628,"train/train/tensor_act_model_layers_83_self_attn_k_proj/mean":0.03509521484375,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/norm":0.01218775180409807,"train/train/tensor_act_model_layers_54_self_attn_v_proj/std":0.34130969272485717,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/max_abs":0.150390625,"train/train/tensor_act_model_layers_53_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/norm":4.71875,"train/train/tensor_act_model_layers_47_self_attn_v_proj/mean":0.00548553466796875,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/std":4.274867125505889e-05,"train/train/tensor_act_model_layers_53_mlp_up_proj/mean":-0.0848388671875,"train/train/tensor_act_model_layers_66_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/norm":0.016848577724051215,"train/train/tensor_act_model_layers_5_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_post_attention_layernorm/mean":0.0129241943359375,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/norm":0.010056039800752743,"train/train/tensor_act_model_layers_68_post_attention_layernorm/mean":0.0419921875,"train/train/tensor_act_model_layers_29_mlp/max_abs":0.98046875,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/mean":-3.1461240723729134e-08,"train/train/layer_model_layers_60/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/max_abs":0.1259765625,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_10_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/mean":2.6135239750146866e-08,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/max_abs":0.11083984375,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/std":0.052978515625,"train/train/layer__model_layers_77/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/max_abs":0.1376953125,"train/train/tensor_act_model_layers_70_self_attn_o_proj/mean":-0.002994537353515625,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/max_abs":0.140625,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_post_attention_layernorm/mean":0.06109619140625,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/std":3.386463907983374e-05,"train/train/layer_model_layers_18/act/norm":13934.747762524217,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/mean":-6.565824151039124e-07,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_64/grad/norm":0.06303996661194479,"train/train/layer__model_layers_51/param/max_abs":1,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_6/grad/norm":0.04084197028419201,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/max_abs":0.181640625,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/mean":-0.0003223419189453125,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_o_proj/std":0.13672054807936632,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/norm":0.005841924590076506,"train/train/tensor_act_model_layers_40_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_43_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54/mean":0.0684814453125,"train/train/tensor_act_model_layers_11_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/mean":3.4955155570060015e-07,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/std":2.0441941771630366e-05,"train/train/tensor_act_model_layers_6_input_layernorm/mean":0.0009660720825195312,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/norm":0.021730830037159823,"train/train/tensor_act_model_layers_82_input_layernorm/mean":0.056396484375,"train/train/tensor_act_model_layers_54_input_layernorm/mean":0.04779052734375,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_mlp_up_proj/std":0.4477550377618268,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/std":7.095206474194892e-05,"train/train/tensor_act_model_layers_20_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_12/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/norm":0.0053097297303538364,"train/train/layer_model_layers_47/grad/std":6.609428438397878e-05,"train/train/tensor_act_model_layers_83_input_layernorm/mean":0.0628662109375,"train/train/tensor_act_model_layers_51_self_attn_q_proj/std":0.8808616312509809,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_o_proj/norm":755.8110264089598,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_33/param/std":0.04998731433972883,"train/train/tensor_act_model_layers_45_post_attention_layernorm/mean":0.068603515625,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/norm":0.019756637912406654,"train/train/tensor_act_model_layers_84_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_7/act/norm":17673.052270815166,"train/train/tensor_act_model_layers_44_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_11_input_layernorm/norm":5792.609008794993,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_86/grad/max_abs":0.0029144287109375,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/mean":0.00023746490478515625,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/mean":-4.5495107769966125e-07,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/std":5.068502224511829e-05,"train/train/tensor_act_model_layers_32_self_attn/mean":-2.8818845748901367e-05,"train/train/tensor_act_model_layers_7_post_attention_layernorm/norm":5792.60241699644,"train/train/layer__model_layers_50/param/norm":21.51226031243579,"train/train/tensor_act_model_layers_57_self_attn_v_proj/std":0.4238282056688361,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/max_abs":0.13671875,"train/train/tensor_act_model_layers_78_self_attn_o_proj/max_abs":1.8671875,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/max_abs":0.130859375,"train/train/tensor_act_model_layers_63_post_attention_layernorm/norm":5792.603271487097,"train/train/tensor_act_model_layers_71_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/mean":-0.00028228759765625,"train/train/tensor_act_lm_head/frac_near_dtype_limit":0,"train/train/layer_model_layers_15/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn/std":0.08510432143573667,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn_q_proj/norm":5493.008499031177,"train/train/tensor_act_model_layers_41/std":1.2988445997985918,"train/train/tensor_act_model_layers_71_mlp/std":0.15283282528225764,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/std":0.0296630859375,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/norm":3.265625,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/norm":0.012406849981160444,"train/train/tensor_act_model_layers_41_post_attention_layernorm/mean":0.0614013671875,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/norm":5.125,"train/train/tensor_act_model_layers_82_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/norm":5.78125,"train/train/tensor_param_model_layers_26_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/std":0.04150390625,"train/train/tensor_act_model_layers_28_mlp_down_proj/norm":284.54347916169047,"train/train/tensor_act_model_layers_7_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/global/grad/max_abs":0.01361083984375,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/max_abs":0.000888824462890625,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp/norm":283.46965340995047,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/std":0.000129713024800763,"train/train/tensor_act_model_layers_90_self_attn/norm":898.9729478617083,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/norm":0.013531474869249356,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/norm":0.002587051075481949,"train/train/layer_model_layers_39/grad/norm":0.03586848911532497,"train/train/tensor_act_model_layers_43/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_79/act/norm":16532.63977359766,"train/train/tensor_act_model_layers_65_mlp_down_proj/max_abs":1.234375,"train/train/tensor_act_model_layers_8/max_abs":18.5,"train/train/tensor_act_model_layers_6_self_attn_o_proj/norm":562.2322288928617,"train/train/tensor_act_model_layers_82_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/norm":0.005073610463144384,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_down_proj/max_abs":0.5859375,"train/train/layer_model_layers_56/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/mean":-1.1431984603404999e-07,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/mean":-5.3783878684043884e-08,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/mean":0.0004634857177734375,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/mean":2.60770320892334e-08,"train/train/tensor_act_model_layers_8_self_attn_o_proj/max_abs":0.59375,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/max_abs":0.000308990478515625,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/norm":0.04315336621349875,"train/train/tensor_act_model_layers_58_input_layernorm/max_abs":5.71875,"train/train/tensor_act_model_layers_80_post_attention_layernorm/norm":5792.609130865719,"train/train/tensor_param_model_layers_75_input_layernorm_weight/mean":1,"train/train/layer_model_layers_54/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/norm":6.8125,"train/train/tensor_act_model_layers_10_input_layernorm/std":1.0000000045110937,"train/train/tensor_act_model_layers_1_mlp_up_proj/mean":-0.05328369140625,"train/train/tensor_act_model_layers_10_self_attn_v_proj/std":0.30078145752762625,"train/train/layer_model_layers_29/grad/norm":0.03681817518877549,"train/train/tensor_act_model_layers_50_input_layernorm/mean":0.05084228515625,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/norm":4.15625,"train/train/tensor_act_model_layers_50_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/max_abs":0.1064453125,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_17_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_66/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/max_abs":0.10205078125,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/mean":3.309105522930622e-08,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_post_attention_layernorm/max_abs":5.40625,"train/train/tensor_act_model_layers_65_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_58/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/mean":0.000759124755859375,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/std":4.9972291099627613e-05,"train/train/tensor_act_model_layers_77_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_q_proj/max_abs":5.8125,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/std":0.0240478515625,"train/train/layer_model_layers_25/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/mean":-1.9266735762357712e-07,"train/train/tensor_act_model_layers_51_self_attn_q_proj/norm":5114.251042702868,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/std":9.599080197062525e-05,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/max_abs":0.1103515625,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43/max_abs":16.5,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/max_abs":0.126953125,"train/train/tensor_act_model_layers_47_self_attn_o_proj/std":0.17114460530862655,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/std":3.1772490574918524e-05,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/std":0.0242919921875,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/norm":0.029588267470068954,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/max_abs":0.000629425048828125,"train/train/tensor_act_model_layers_10_self_attn/mean":-0.000850677490234375,"train/train/tensor_param_model_layers_87_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_63_input_layernorm/std":1.000000081956383,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/max_abs":0.00066375732421875,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/std":0.042236328125,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/max_abs":0.0002994537353515625,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/norm":0.022035973766183183,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/mean":-9.72747802734375e-05,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/max_abs":0.1845703125,"train/train/tensor_act_model_layers_85_self_attn/norm":807.6520046408407,"train/train/tensor_act_model_layers_41_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/max_abs":0.000431060791015625,"train/train/tensor_act_model_layers_32_self_attn_v_proj/norm":1600.752091938115,"train/train/tensor_act_model_layers_73_mlp/frac_near_dtype_limit":0,"train/train/layer__model_layers_46/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/norm":5.125,"train/train/tensor_act_model_layers_38_input_layernorm/norm":5792.606567390163,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/std":3.276114397664286e-05,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/max_abs":0.20703125,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/norm":7.09375,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/std":0.041015625,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/norm":0.011908661290296125,"train/train/tensor_act_model_layers_82/mean":0.1082763671875,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/mean":-0.0002841949462890625,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/mean":1.3728276826441288e-07,"train/train/tensor_act_model_layers_45_input_layernorm/norm":5792.6109619186755,"train/train/tensor_param_model_layers_76_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_15_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn_q_proj/norm":5656.414935315061,"train/train/tensor_act_model_layers_16_self_attn_v_proj/mean":-0.0005960464477539062,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/norm":8.9375,"train/train/tensor_act_model_layers_33_post_attention_layernorm/mean":0.0650634765625,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/max_abs":0.00011730194091796875,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/std":0.03564453125,"train/train/tensor_act_model_layers_32_mlp/std":0.0451660233574938,"train/train/tensor_act_model_layers_37_self_attn_o_proj/std":0.05114804200835862,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/max_abs":0.19140625,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_v_proj/max_abs":2.671875,"train/train/tensor_act_model_layers_64_self_attn_k_proj/mean":0.035400390625,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/max_abs":0.0001506805419921875,"train/train/tensor_act_model_layers_8_self_attn_v_proj/mean":-0.00763702392578125,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_50_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_48/grad/norm":0.04209173518518737,"train/train/tensor_act_model_layers_0_mlp_down_proj/std":1.4531565478192037,"train/train/tensor_act_model_layers_81/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_k_proj/mean":-0.0247802734375,"train/train/tensor_act_model_layers_13_mlp_up_proj/norm":2242.066636163669,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/std":0.053466796875,"train/train/tensor_act_model_layers_3_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/std":0.0001214727682098318,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/max_abs":0.000774383544921875,"train/train/tensor_act_model_layers_22/frac_near_user_limit":0,"train/train/layer_model_layers_21/act/mean":-0.00747827383188101,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/max_abs":0.265625,"train/train/tensor_param_model_layers_7_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/std":0.046142578125,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/std":5.4976925626192864e-05,"train/train/tensor_act_model_layers_38_self_attn/std":0.1242678663699174,"train/train/tensor_act_model_layers_56_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_post_attention_layernorm/max_abs":6.0625,"train/train/tensor_act_model_layers_70_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33/std":1.3008106446023608,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/max_abs":0.0002193450927734375,"train/train/tensor_param_model_layers_70_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/mean":0.0010986328125,"train/train/layer_model_layers_72/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_58/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/std":9.55458241201768e-05,"train/train/layer_model_layers_53/act/mean":-0.0009266779972956731,"train/train/layer__model_layers_1/param/max_abs":1,"train/train/tensor_act_model_layers_35_mlp_up_proj/norm":3042.3214710769407,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/std":0.0279541015625,"train/train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/std":4.7074668726716624e-05,"train/train/tensor_act_model_layers_38_post_attention_layernorm/std":0.9960939519545406,"train/train/tensor_act_model_layers_53/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp_down_proj/max_abs":1.1015625,"train/train/tensor_act_model_layers_31_self_attn/norm":294.3296823336847,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/std":1.555979162509807e-05,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/max_abs":0.00011444091796875,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_k_proj/norm":4715.804521325147,"train/train/tensor_act_model_layers_86_mlp_up_proj/std":0.6054692914406601,"train/train/tensor_act_model_layers_56_post_attention_layernorm/std":1.0000001396983766,"train/train/tensor_act_model_layers_71_self_attn_q_proj/std":0.8222679960155508,"train/train/tensor_act_model_layers_31_self_attn_o_proj/mean":-0.0008831024169921875,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_input_layernorm/std":1.0000011860385953,"train/train/tensor_act_model_layers_54_self_attn/std":0.05786356790851186,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/max_abs":0.00024127960205078125,"train/train/tensor_act_model_layers_61_input_layernorm/norm":5792.606323245237,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/std":5.693493432673167e-05,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/std":5.472345475593096e-05,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/mean":0.0004177093505859375,"train/train/tensor_act_model_layers_29_self_attn_k_proj/std":0.7929688322132988,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/std":0.0301513671875,"train/train/layer__model_layers_27/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/mean":-3.0249357223510742e-05,"train/train/tensor_act_model_layers_40_self_attn_v_proj/norm":2187.289299619383,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/max_abs":0.00012063980102539062,"train/train/tensor_act_model_layers_34_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/max_abs":0.140625,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/std":7.190959544055063e-05,"train/train/tensor_act_model_layers_59_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/std":3.8094416150191325e-05,"train/train/tensor_act_model_layers_26_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_o_proj/mean":0.0002055540680885315,"eval/loss":1.8617645502090454,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/max_abs":0.000514984130859375,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/std":0.00012780885726472428,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/std":4.8137419845914026e-05,"train/train/tensor_act_model_layers_44_self_attn_q_proj/norm":5491.670192433938,"train/train/layer__model_layers_44/param/mean":0.0015361149113933307,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/norm":6.65625,"train/train/tensor_act_model_layers_15_mlp_up_proj/std":0.22753916073252642,"train/train/tensor_act_model_layers_86_self_attn/max_abs":2.71875,"train/train/layer__model_layers_35/param/std":0.050796343952218985,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_post_attention_layernorm/max_abs":6.09375,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/norm":0.035961673733179475,"train/train/tensor_act_model_layers_74_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_44/grad/mean":-4.05273517933725e-07,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/mean":1.3317912817001343e-06,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/norm":0.027747511103423602,"train/train/tensor_act_model_layers_10_self_attn_q_proj/norm":6506.856107360179,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_84_self_attn_v_proj/std":0.4355469321990724,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/norm":6.84375,"train/train/tensor_act_model_layers_64_self_attn_q_proj/std":0.9941438752628997,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_post_attention_layernorm/mean":-0.00522613525390625,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_o_proj/std":0.13061775250770502,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/std":0.02685546875,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/mean":0.0003032684326171875,"train/train/tensor_act_model_layers_78_self_attn_q_proj/max_abs":6.25,"train/train/layer__model_layers_16/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_post_attention_layernorm/norm":5792.613281255463,"train/train/tensor_act_model_layers_41_self_attn_v_proj/norm":1785.494370843625,"train/train/tensor_act_model_layers_57/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/mean":0.00013637542724609375,"train/train/tensor_act_model_layers_39_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_v_proj/std":0.4648438200448438,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/norm":0.006378502358184442,"train/train/tensor_act_model_layers_63_self_attn/norm":755.8110264089598,"train/train/tensor_act_model_layers_19_mlp_down_proj/norm":222.34625983155243,"train/train/tensor_act_model_layers_31_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_51_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/max_abs":0.00145721435546875,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/mean":7.152557373046875e-05,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_post_attention_layernorm/mean":0.0831298828125,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/std":0.0252685546875,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/norm":5.46875,"train/train/tensor_act_model_layers_73_self_attn_v_proj/mean":0.0010519027709960938,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/std":6.532942850088739e-05,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/max_abs":0.000713348388671875,"train/train/tensor_param_model_layers_45_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_78/grad/max_abs":0.001495361328125,"train/train/tensor_act_model_layers_35_self_attn/norm":719.8087211490051,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/mean":0.00015544891357421875,"train/train/tensor_act_model_embed_tokens/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_input_layernorm/norm":5792.61010742507,"train/train/tensor_param_model_layers_63_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_17_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_input_layernorm/mean":0.044921875,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/norm":3.546875,"train/train/tensor_act_model_layers_38_self_attn_v_proj/max_abs":2.484375,"train/train/layer__model_layers_25/param/std":0.04979521890365126,"train/train/layer__model_layers_12/param/frac_near_user_limit":0,"train/train/tensor_act_model_embed_tokens/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_post_attention_layernorm/max_abs":6.25,"train/train/tensor_act_model_layers_86_mlp_down_proj/norm":1429.1296013645515,"train/train/tensor_act_model_layers_81_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp/max_abs":1.515625,"train/train/tensor_act_model_layers_59_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn_v_proj/mean":-0.00414276123046875,"train/train/tensor_act_model_layers_49_mlp/norm":465.9405843886969,"train/train/layer_model_layers_62/act/norm":15039.461222524924,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/norm":3.21875,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/std":3.66657109050401e-05,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/norm":2.859375,"train/train/global/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17/norm":7960.625918807501,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/std":6.85222090602087e-05,"train/train/tensor_param_model_layers_61_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_78_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/norm":5.46875,"train/train/tensor_act_model_layers_13_post_attention_layernorm/norm":5792.610229497346,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/max_abs":0.25390625,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/norm":0.023409588406361183,"train/train/tensor_act_model_layers_72_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/std":7.697323581504032e-05,"train/train/layer_model_layers_91/grad/std":0.00012836935604943137,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18/std":1.3711216114071696,"train/train/tensor_act_model_layers_68_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/mean":-1.5974044799804688e-05,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/norm":0.007480137919787043,"train/train/tensor_act_model_layers_82_mlp_down_proj/max_abs":2.328125,"train/train/tensor_act_model_layers_20_mlp_up_proj/std":0.22753919347669246,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/std":0.0001481692345793827,"train/train/tensor_act_model_layers_10_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/norm":497.63441981347637,"train/train/tensor_act_model_layers_34_input_layernorm/max_abs":6.28125,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_57/max_abs":14.5625,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/std":4.5668900771282465e-05,"train/train/tensor_act_model_layers_81_self_attn/norm":1258.459225299919,"train/train/tensor_act_model_layers_40_mlp_up_proj/max_abs":3.390625,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/mean":0.000179290771484375,"train/train/tensor_act_model_layers_91_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_post_attention_layernorm/max_abs":6.0625,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_71/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/std":0.041259765625,"train/train/layer_model_layers_70/act/max_abs":13.4375,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/max_abs":8.0108642578125e-05,"train/train/layer_model_layers_16/grad/std":3.684865338850418e-05,"train/train/tensor_act_model_layers_0_self_attn_k_proj/max_abs":6.5,"train/train/tensor_act_model_layers_85_self_attn_o_proj/max_abs":2.25,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/mean":-0.00030517578125,"train/train/tensor_act_model_layers_27_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/std":2.6659910049747163e-05,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn_o_proj/norm":1231.0956781488853,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/norm":0.017949940331766313,"train/train/tensor_act_model_layers_85_mlp_up_proj/norm":6166.550623580844,"train/train/layer_model_layers_0/act/std":0.959431948114704,"train/train/tensor_act_model_layers_12_self_attn_v_proj/max_abs":1.4453125,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/mean":9.012222290039062e-05,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/max_abs":0.0003719329833984375,"train/train/layer_model_layers_49/grad/mean":-1.0023224814075762e-06,"train/train/tensor_act_model_layers_76_self_attn/std":0.12598226066108384,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/std":4.108033103584092e-05,"train/train/tensor_param_model_layers_29_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/mean":-4.782341420650482e-06,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/mean":-1.0356307029724121e-06,"train/train/tensor_act_model_layers_53_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_37/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55/max_abs":14.875,"train/train/tensor_act_model_layers_22/std":1.3554969737937457,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/mean":-6.076879799365997e-08,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/norm":0.0049526230906754365,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/mean":-2.5203917175531387e-08,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/mean":-3.448221832513809e-07,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/std":0.0322265625,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/max_abs":0.25390625,"train/train/tensor_act_model_layers_43_mlp_down_proj/max_abs":1.09375,"train/train/tensor_param_model_layers_21_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_36_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_input_layernorm/mean":0.0692138671875,"train/train/tensor_act_model_layers_88_post_attention_layernorm/mean":0.0772705078125,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/norm":5.4375,"train/train/tensor_act_model_layers_60/mean":0.06085205078125,"train/train/tensor_act_model_layers_35_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/mean":1.0174699127674103e-07,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/mean":0.000591278076171875,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/std":1.6551417886374604e-05,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/std":0.023193359375,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/norm":5,"train/train/tensor_act_model_layers_4_self_attn_k_proj/norm":6374.149442562751,"train/train/tensor_act_model_layers_61_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_41/grad/norm":0.0340792282305843,"train/train/tensor_act_model_layers_82_self_attn/mean":-0.000946044921875,"train/train/tensor_act_model_layers_58_self_attn_o_proj/std":0.08510432143573667,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/max_abs":0.00113677978515625,"train/train/tensor_act_model_layers_72_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp/max_abs":0.466796875,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp_up_proj/max_abs":3.140625,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/mean":-4.744529724121094e-05,"train/train/tensor_act_model_layers_53_mlp_down_proj/norm":518.2545092485765,"train/train/tensor_act_model_layers_21_self_attn_v_proj/std":0.3115250028186743,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/std":3.782520942055395e-05,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/max_abs":0.267578125,"train/train/tensor_act_model_layers_7_mlp_down_proj/norm":343.9711612161347,"train/train/tensor_act_model_layers_28_self_attn_v_proj/max_abs":2.015625,"train/train/tensor_act_model_layers_83_self_attn_o_proj/mean":0.001979827880859375,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_41/grad/std":4.210039873989049e-05,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/mean":-2.0503997802734375e-05,"train/train/global/param/std":0.05651444415048722,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/max_abs":0.002105712890625,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/norm":0.0052676646105632875,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/max_abs":0.259765625,"train/train/tensor_act_model_layers_32_self_attn_v_proj/std":0.27588028916124363,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/max_abs":0.000232696533203125,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/norm":0.0006267838985010861,"train/train/layer_model_layers_67/grad/mean":3.816649991896893e-07,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn/max_abs":0.953125,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/mean":-8.865026757121086e-08,"train/train/tensor_act_model_layers_1_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/std":1.6671943133417003e-05,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/mean":-0.00113677978515625,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/std":0.04345703125,"train/train/layer_model_layers_10/grad/max_abs":0.0016021728515625,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/max_abs":0.23046875,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/mean":0.0003452301025390625,"train/train/tensor_act_model_layers_61_self_attn_k_proj/max_abs":4.71875,"train/train/tensor_act_model_layers_73_self_attn_k_proj/norm":4536.17631723189,"train/train/tensor_act_model_layers_40_self_attn_q_proj/max_abs":5.875,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/mean":0.0002880096435546875,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_input_layernorm/std":1.0000001098960578,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/mean":-0.00124359130859375,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/norm":0.025212617594152022,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_o_proj/norm":447.84223212527223,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25/max_abs":17.75,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/max_abs":0.000247955322265625,"train/train/tensor_act_model_layers_89_self_attn_o_proj/norm":1783.3119747617445,"train/train/tensor_act_model_layers_70_mlp_down_proj/mean":-0.0015506744384765625,"train/train/tensor_act_model_layers_54_self_attn_k_proj/max_abs":4.65625,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/mean":1.9073486328125e-05,"train/train/tensor_act_model_layers_56/max_abs":14.625,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/mean":4.414469003677368e-07,"train/train/layer_model_layers_29/grad/mean":-5.352876385944124e-07,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/norm":0.047143169067638956,"train/train/layer_model_layers_23/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp_down_proj/norm":696.4606801217087,"train/train/tensor_act_model_layers_3/max_abs":18.5,"train/train/layer__model_layers_41/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/mean":0.000545501708984375,"train/train/layer__model_layers_2/param/norm":19.20518449086405,"train/train/tensor_param_model_layers_1_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_v_proj/std":0.36132816473349105,"train/train/tensor_act_model_layers_67_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/max_abs":0.6015625,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/max_abs":0.142578125,"train/train/tensor_act_model_layers_44_self_attn/norm":533.673284835028,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/std":4.3459484328028e-05,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/norm":0.0032491100329550535,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/norm":0.008069201994025138,"train/train/layer__model_layers_3/param/norm":19.76755178355744,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/std":0.0218505859375,"train/train/tensor_act_model_layers_67_post_attention_layernorm/mean":0.0377197265625,"train/train/tensor_act_model_layers_10_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/max_abs":0.00015926361083984375,"train/train/tensor_act_model_layers_58_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/mean":5.3085386753082275e-08,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/max_abs":9.918212890625e-05,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/norm":0.0015753452722740307,"train/train/tensor_act_model_layers_64_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/max_abs":0.00045013427734375,"train/train/layer_model_layers_71/grad/max_abs":0.0009307861328125,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/max_abs":0.00066375732421875,"train/train/tensor_act_model_layers_24_self_attn_v_proj/max_abs":1.9453125,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/mean":7.776543498039246e-08,"train/train/tensor_act_model_layers_51/std":1.2949382657240718,"train/train/tensor_act_model_layers_40_self_attn_k_proj/std":0.8349630344654543,"train/train/layer_model_layers_93/act/std":1.3383544041746744,"train/train/tensor_act_model_layers_3_input_layernorm/mean":0.019989013671875,"train/train/tensor_act_model_layers_70_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_norm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/std":3.824687171401027e-05,"train/train/tensor_act_model_layers_64_self_attn/mean":-0.0017452239990234375,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/mean":1.2926757335662842e-06,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/norm":0.009630459936736275,"train/train/tensor_act_model_layers_63_self_attn/std":0.13061775250770502,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/layer_model_layers_0/grad/mean":-2.6291948212848253e-06,"train/train/layer__model_layers_56/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/norm":0.005092550841575362,"train/train/layer_model_layers_53/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/std":6.431615326066655e-05,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/max_abs":0.00109100341796875,"train/train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp_down_proj/max_abs":1.5,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_9_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_up_proj/std":0.3398438376941787,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/std":0.00012183271042014564,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_mlp/mean":0.00476837158203125,"train/train/layer_model_layers_77/act/mean":-0.0003136579806988056,"train/train/tensor_act_model_layers_66_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_81_mlp_up_proj/norm":6013.202664725769,"train/train/tensor_act_model_layers_90_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/norm":5.28125,"train/train/tensor_act_model_layers_3_post_attention_layernorm/norm":5792.607055673553,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/std":8.267120584977433e-05,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/max_abs":0.0002040863037109375,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/mean":3.762543201446533e-07,"train/train/tensor_act_model_layers_7_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/norm":0.03748691648491507,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn_v_proj/mean":-0.002368927001953125,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/norm":5.3125,"train/train/layer_model_layers_35/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_post_attention_layernorm/norm":5792.606933595222,"train/train/tensor_act_model_layers_38_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_post_attention_layernorm/norm":5792.610595707493,"train/train/layer__model_layers_40/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_post_attention_layernorm/norm":5792.607910157084,"train/train/layer_model_layers_67/grad/norm":0.05884391681589697,"train/train/tensor_param_model_layers_22_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_62_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/mean":-0.00029754638671875,"train/train/tensor_act_model_layers_5_mlp_down_proj/mean":-0.002838134765625,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_49/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/std":0.042236328125,"train/train/tensor_act_model_layers_1_self_attn_q_proj/norm":8044.530510496891,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/std":2.815077386268755e-05,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/mean":-5.58607280254364e-06,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6/max_abs":18.75,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/std":5.0849912601112505e-05,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_2_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_86_self_attn_q_proj/mean":0.04254150390625,"train/train/tensor_act_model_layers_2_self_attn_o_proj/mean":-0.00067901611328125,"train/train/layer_model_layers_68/act/max_abs":13.1875,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_61_post_attention_layernorm/max_abs":5.96875,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/max_abs":0.00010585784912109375,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/std":0.00011848272727289888,"train/train/tensor_act_model_layers_46_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/std":8.036104814102122e-05,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_93/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/mean":-3.695487976074219e-05,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/max_abs":0.000690460205078125,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/mean":1.775333657860756e-09,"train/train/tensor_act_model_layers_38_self_attn_q_proj/mean":0.03863525390625,"train/train/tensor_act_model_layers_40_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_post_attention_layernorm/max_abs":5.96875,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/std":0.0546875,"train/train/tensor_act_model_layers_33_self_attn_v_proj/std":0.2978533567089106,"train/train/tensor_act_model_layers_70_self_attn/std":0.26027066890145434,"train/train/tensor_act_model_layers_79_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_21/grad/max_abs":0.0018768310546875,"train/train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp_up_proj/norm":4525.33510720393,"train/train/tensor_act_model_layers_42_mlp_up_proj/max_abs":2.890625,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/max_abs":0.140625,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/std":0.039306640625,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/norm":0.009630846752683761,"train/train/tensor_act_model_layers_4_mlp_down_proj/std":0.06591796875,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/norm":4.96875,"train/train/tensor_act_model_layers_31_self_attn_q_proj/std":0.9687513689831321,"train/train/tensor_act_model_layers_34_self_attn/std":0.08667096912647912,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_k_proj/norm":4660.636121871704,"train/train/layer_model_layers_91/grad/mean":3.848730494935129e-07,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/mean":1.8358230590820312e-05,"train/train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_input_layernorm/mean":0.07568359375,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn/max_abs":1.8515625,"train/train/layer_model_layers_57/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_input_layernorm/norm":5792.605346692513,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/max_abs":0.2216796875,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/max_abs":0.20703125,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/std":0.0255126953125,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_down_proj/norm":558.7376441442592,"train/train/tensor_act_model_layers_22_self_attn_v_proj/std":0.3227550691246056,"train/train/tensor_act_model_layers_60_input_layernorm/std":1.0000000651925782,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/mean":-0.0003490447998046875,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_6/grad/mean":-2.650157570281006e-07,"train/train/tensor_act_model_layers_85_self_attn_k_proj/mean":-0.016754150390625,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_o_proj/max_abs":3.5625,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/norm":5.25,"train/train/layer_model_layers_59/act/mean":-0.012381333571213942,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_2_self_attn_k_proj/mean":-0.0506591796875,"train/train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp_up_proj/max_abs":2.828125,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/norm":0.0007280702260101881,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/norm":0.002392189325155039,"train/train/layer_model_layers_84/act/std":0.8248708860955685,"train/train/tensor_act_model_layers_33_self_attn_o_proj/norm":275.26248700016043,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/norm":6.375,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/mean":-0.000698089599609375,"train/train/tensor_param_model_layers_42_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_k_proj/max_abs":4.21875,"train/train/tensor_act_model_layers_31_self_attn_k_proj/mean":0.01568603515625,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/norm":0.0159548280051682,"train/train/tensor_act_model_layers_79_self_attn_q_proj/max_abs":6.15625,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_q_proj/std":0.9570314266243597,"train/train/tensor_act_model_layers_75_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_up_proj/max_abs":2.75,"train/train/layer_model_layers_17/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn/mean":-0.001384735107421875,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_82_self_attn_o_proj/max_abs":1.9609375,"train/train/tensor_act_model_layers_77_self_attn_k_proj/max_abs":5.875,"train/train/tensor_act_model_layers_12_self_attn_v_proj/norm":1349.7732985964547,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/norm":2.65625,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/std":0.0001299477648686743,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/norm":0.0013141618238643891,"train/train/tensor_param_model_layers_34_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_17_mlp_up_proj/mean":-0.05322265625,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_mlp/std":0.12048520238544266,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/std":7.754239185070194e-05,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/mean":2.2264430299401283e-07,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/max_abs":0.0006256103515625,"train/train/layer__model_layers_11/param/std":0.04836446008345068,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/std":0.025390625,"train/train/tensor_act_model_layers_47_self_attn/max_abs":1.4609375,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_mlp_up_proj/mean":-0.0828857421875,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_92/act/mean":0.0031217428354116585,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/std":0.0400390625,"train/train/tensor_act_model_layers_1_self_attn/std":0.09350619070153761,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_49_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_q_proj/mean":0.1044921875,"train/train/tensor_act_model_layers_6_self_attn_q_proj/mean":0.0205078125,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/std":0.026611328125,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/norm":6.84375,"train/train/tensor_param_model_layers_13_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_self_attn_v_proj/std":0.2436530585492926,"train/train/layer_model_layers_70/act/mean":-0.01564818162184495,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn_o_proj/norm":690.7827898320194,"train/train/tensor_act_model_layers_11_self_attn_q_proj/std":1.0703125904946393,"train/train/tensor_act_model_layers_20_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_65/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/std":9.642636729990546e-05,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/max_abs":0.0005035400390625,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/mean":-0.00051116943359375,"train/train/tensor_act_model_layers_64_self_attn_o_proj/mean":-0.0017452239990234375,"train/train/tensor_param_model_layers_72_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/norm":0.01054491625363687,"train/train/tensor_act_model_layers_55/std":1.320324207852161,"train/train/tensor_act_model_layers_40_self_attn_o_proj/max_abs":1.1875,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn/mean":-0.00296783447265625,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/std":2.031110006526671e-05,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/mean":-2.0265579223632812e-06,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/norm":0.0011064858046190276,"train/train/tensor_act_model_layers_89_self_attn/std":0.30763140751502616,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/norm":0.026734652555129638,"train/train/tensor_act_model_layers_10_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_v_proj/norm":1830.530855228877,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/mean":-7.470225682482123e-08,"train/train/tensor_act_model_layers_21_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_o_proj/mean":-0.0008916854858398438,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/max_abs":0.00060272216796875,"train/train/tensor_act_model_layers_65_mlp/norm":737.1329428876805,"train/train/tensor_act_model_layers_5_mlp/max_abs":0.515625,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/norm":6.40625,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_up_proj/mean":-0.05670166015625,"train/train/tensor_act_model_layers_71/mean":0.03375244140625,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/mean":-8.165836334228516e-06,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/std":2.4411583214979407e-05,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/std":6.640012708534531e-05,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/mean":6.151199340820312e-05,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/mean":-9.709969162940979e-06,"train/train/tensor_act_model_layers_34_mlp/mean":0.0080413818359375,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/max_abs":0.240234375,"train/train/tensor_act_model_layers_39_self_attn_o_proj/norm":359.54123512118264,"train/train/tensor_act_model_layers_34_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/std":0.00014046782129582404,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/norm":0.0030508326081655447,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/max_abs":0.00139617919921875,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/norm":6.40625,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/mean":0.0003299713134765625,"train/train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/std":0.000116604225509123,"train/train/tensor_act_model_layers_73_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/max_abs":0.000698089599609375,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/max_abs":0.193359375,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/std":0.0294189453125,"train/train/layer_model_layers_2/act/std":0.7600054486092788,"train/train/tensor_act_model_layers_88/max_abs":12.3125,"train/train/tensor_act_model_layers_19_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/mean":-1.9953586161136627e-07,"train/train/tensor_act_model_layers_44_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/std":1.4463113317351015e-05,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/std":0.00010915881549135672,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/std":0.052978515625,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/norm":4.71875,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/mean":-2.0172446966171265e-06,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/mean":-2.000480890274048e-06,"train/train/tensor_act_model_layers_56_self_attn_k_proj/std":0.7763718181579222,"train/train/tensor_act_model_layers_48_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp/max_abs":4.71875,"train/train/tensor_act_model_layers_76_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/max_abs":0.1279296875,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/norm":0.0012924056970075738,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/max_abs":0.00018978118896484375,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/max_abs":0.1220703125,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/max_abs":9.012222290039062e-05,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/max_abs":0.00011014938354492188,"train/train/tensor_act_model_layers_59_self_attn_q_proj/std":0.862306447421532,"train/train/tensor_act_model_layers_58_self_attn/mean":-0.0012683868408203125,"train/train/tensor_act_model_layers_5/mean":0.001842498779296875,"train/train/tensor_act_model_layers_7_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_k_proj/norm":5650.244450354145,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/norm":0.02384600671910341,"train/train/tensor_act_model_layers_84_input_layernorm/std":0.9960958144222632,"train/train/tensor_act_model_layers_87_input_layernorm/norm":5792.611938477483,"train/train/tensor_act_model_layers_69_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_42/param/max_abs":1,"train/train/total_time_seconds":2608.5188849791884,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_post_attention_layernorm/mean":0.046875,"train/train/layer__model_layers_30/param/max_abs":1,"train/train/layer_model_layers_24/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/norm":0.004479565718159379,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_72/grad/norm":0.05921144206464787,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/norm":0.02384569427142626,"train/train/tensor_act_model_layers_27_mlp/std":0.04321289803348628,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/std":0.0001745280176134684,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/norm":0.022798979815655995,"train/train/tensor_act_model_layers_82_mlp_down_proj/norm":1306.9650487676554,"train/train/tensor_act_model_layers_86_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70/mean":0.03546142578125,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/norm":0.03977263593043948,"train/train/tensor_act_model_layers_15_mlp/norm":258.7885784510314,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/std":4.915089954725081e-05,"train/train/tensor_act_model_layers_90/max_abs":12.9375,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/mean":-0.0001468658447265625,"train/train/layer_model_layers_18/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_36/act/std":0.6731045492503508,"train/train/tensor_act_model_layers_91_mlp/mean":-0.009796142578125,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_79/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/max_abs":0.00016307830810546875,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/mean":0.0001430511474609375,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp/mean":0.001880645751953125,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/max_abs":0.0005035400390625,"train/train/tensor_act_model_layers_81/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_4/act/norm":15342.662595333373,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/norm":0.015521091425966515,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_66/grad/norm":0.06407017284803296,"train/train/tensor_act_model_layers_42_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/mean":0.0005340576171875,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_65_mlp/max_abs":1.234375,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/std":4.790105612906847e-05,"train/train/tensor_act_model_layers_88_self_attn_q_proj/mean":0.044921875,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/max_abs":0.0004730224609375,"train/train/tensor_act_model_layers_0_input_layernorm/norm":5792.457153412227,"train/train/tensor_act_model_layers_22_post_attention_layernorm/norm":5792.614624026353,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/max_abs":0.000415802001953125,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_22/param/mean":0.0016206571724783055,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/mean":-0.0003261566162109375,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/mean":-0.0007781982421875,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/std":7.463168866980159e-05,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/mean":-5.333276931196451e-08,"train/train/tensor_act_model_layers_15_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/mean":-0.00032806396484375,"train/train/tensor_param_model_layers_83_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_26_post_attention_layernorm/mean":0.0333251953125,"train/train/tensor_act_model_layers_43_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/max_abs":0.12890625,"train/train/layer_model_layers_28/grad/mean":-6.125080464596682e-07,"train/train/layer__model_layers_7/param/mean":0.0016048947660115876,"train/train/tensor_act_model_layers_8_self_attn/mean":0.0007352828979492188,"train/train/tensor_act_model_layers_36_self_attn_k_proj/max_abs":5.65625,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/mean":-6.628036499023438e-05,"train/train/tensor_act_model_layers_68_mlp/norm":807.1160007737263,"train/train/tensor_act_model_layers_8_self_attn_o_proj/std":0.07007054894830737,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/norm":0.029105963150226254,"train/train/layer_model_layers_25/act/frac_near_user_limit":0,"train/train/layer__model_layers_81/param/std":0.059092307379107624,"train/loss":7.419364166259766,"train/train/tensor_act_model_layers_18/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_v_proj/norm":1477.3706646312298,"train/train/layer_model_layers_76/act/norm":15564.20506512158,"train/train/tensor_act_model_layers_63_self_attn/max_abs":1.8125,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/max_abs":0.000244140625,"train/train/tensor_act_model_layers_90_self_attn/max_abs":2.59375,"train/train/tensor_act_model_layers_89_self_attn_q_proj/max_abs":6.3125,"train/train/tensor_param_model_layers_33_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69/mean":0.0400390625,"train/train/layer__model_layers_5/param/std":0.04839515469434547,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_72/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp_up_proj/std":0.5224638323837366,"train/train/tensor_act_model_layers_70_self_attn_k_proj/mean":0.0018520355224609375,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/norm":0.001773944974333981,"train/train/tensor_act_model_layers_6/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/std":2.9749484269950464e-05,"train/train/layer_model_layers_89/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/max_abs":0.0016632080078125,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_self_attn_o_proj/max_abs":2.40625,"train/train/tensor_act_model_layers_83_self_attn/std":0.1840827039354085,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/norm":6.0625,"train/train/tensor_act_model_layers_3_self_attn_q_proj/std":1.6171875276427337,"train/train/tensor_act_model_layers_5_input_layernorm/std":1.0000000160289344,"train/train/tensor_act_model_layers_56_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_post_attention_layernorm/mean":0.018707275390625,"train/train/tensor_act_model_layers_82_self_attn_q_proj/max_abs":7.0625,"train/train/tensor_act_model_layers_41_mlp_down_proj/norm":385.94885037260656,"train/train/tensor_act_model_layers_45_self_attn_k_proj/norm":4708.107070143837,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_post_attention_layernorm/std":1.0000001937150769,"train/train/tensor_act_model_layers_14_self_attn_o_proj/std":0.05133229969242981,"train/train/tensor_act_model_layers_16_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_k_proj/max_abs":5.625,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_55_self_attn_k_proj/std":0.9062500635226204,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/std":0.04541015625,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_44_self_attn_v_proj/max_abs":2.453125,"train/train/tensor_act_model_layers_39_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/norm":5.25,"train/train/tensor_act_model_layers_53/max_abs":15.3125,"train/train/tensor_act_model_layers_32_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/norm":0.0037793621037889137,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/max_abs":8.535385131835938e-05,"train/train/tensor_act_model_layers_73_mlp/std":0.16455338756183835,"train/train/tensor_act_model_layers_19_post_attention_layernorm/norm":5792.610961915044,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/mean":-4.135072231292725e-06,"train/train/tensor_act_model_layers_30_self_attn/norm":636.1683186661928,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/norm":0.0006089970340385671,"train/train/tensor_act_model_layers_72_self_attn_q_proj/norm":4977.16250616254,"train/train/layer_model_layers_57/grad/mean":-7.004918771768323e-07,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/max_abs":0.255859375,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/mean":1.442531356588006e-07,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_54_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_mlp_down_proj/std":0.0827640414581374,"train/train/layer_model_layers_10/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/max_abs":0.000568389892578125,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/max_abs":0.2421875,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_down_proj/mean":-0.0079803466796875,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_q_proj/std":0.9921924546823918,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/max_abs":0.224609375,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp_up_proj/norm":4450.057649595707,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/std":1.2509951428509734e-05,"train/train/tensor_param_model_layers_68_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/max_abs":0.1279296875,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/mean":3.850436769425869e-08,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/mean":-3.425520844757557e-08,"train/train/tensor_param_model_layers_88_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/mean":-4.363059997558594e-05,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/max_abs":0.232421875,"train/train/tensor_act_model_layers_66_mlp_down_proj/norm":728.174479721167,"train/train/tensor_act_model_layers_42_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/norm":0.0024639519137504796,"train/train/tensor_act_model_layers_85_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_9/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_25_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_20_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_87/param/max_abs":1,"train/train/tensor_act_model_layers_92_self_attn/mean":-0.006683349609375,"train/train/tensor_act_model_layers_43_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/mean":-1.0301937436452135e-07,"train/train/tensor_act_model_layers_17_input_layernorm/max_abs":6.03125,"train/train/tensor_act_model_layers_77_mlp_down_proj/std":0.17285159213394513,"train/train/tensor_act_model_layers_4_self_attn_o_proj/mean":0.0008726119995117188,"train/train/tensor_param_model_layers_12_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/mean":-0.00042247772216796875,"train/train/tensor_act_model_layers_57_mlp_down_proj/norm":573.196105520505,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/norm":5.28125,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/std":0.00011241852894619822,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/norm":0.01623971063132939,"train/train/layer_model_layers_90/grad/std":0.0001185716509789769,"train/train/tensor_act_model_layers_87_post_attention_layernorm/mean":0.0772705078125,"train/train/tensor_act_model_layers_89_self_attn_q_proj/norm":6290.980219339251,"train/train/tensor_act_model_layers_78/mean":0.08642578125,"train/train/tensor_act_model_layers_58_self_attn_k_proj/std":0.7431673615040285,"train/train/tensor_act_model_layers_81_post_attention_layernorm/norm":5792.603637695584,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/std":0.034423828125,"train/train/tensor_act_model_layers_12_post_attention_layernorm/mean":-0.0011548995971679688,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/max_abs":0.00086212158203125,"train/train/tensor_act_model_layers_79_mlp_down_proj/max_abs":1.6875,"train/train/layer__model_layers_34/param/mean":0.0016206452701467433,"train/train/layer_model_layers_27/act/norm":13896.646304488384,"train/train/tensor_act_model_layers_42_input_layernorm/mean":0.06494140625,"train/train/tensor_act_model_layers_18_self_attn_k_proj/mean":0.0782470703125,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91/mean":0.1279296875,"train/train/tensor_act_model_layers_60_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_post_attention_layernorm/std":1.0000017136320691,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/std":0.0277099609375,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_39/param/norm":20.626071231650613,"train/train/tensor_act_model_layers_55_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/mean":-3.62396240234375e-05,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_v_proj/norm":1804.2309951408947,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/std":0.03955078125,"train/train/tensor_act_model_layers_62_mlp_down_proj/mean":0.0016345977783203125,"train/train/tensor_act_model_layers_66_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/norm":7.59375,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/mean":-2.534128725528717e-06,"train/train/layer__model_layers_78/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_67_self_attn_v_proj/norm":2242.0765115351414,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/std":8.19588265804335e-05,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/norm":0.0011891152386649406,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn/std":0.061401816146291824,"train/train/tensor_act_model_layers_88_self_attn/std":0.38964977718304616,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn/mean":-0.003467559814453125,"train/train/tensor_act_model_layers_54_post_attention_layernorm/max_abs":5.84375,"train/train/tensor_act_model_layers_88_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/max_abs":0.126953125,"train/train/tensor_param_model_layers_47_input_layernorm_weight/max_abs":1,"train/train/layer__model_layers_7/param/max_abs":1,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_down_proj/norm":251.49561028282218,"train/train/tensor_act_model_layers_47_self_attn_k_proj/std":0.9755875570266616,"train/train/tensor_act_model_layers_49_input_layernorm/std":1.0000000745058033,"train/train/tensor_act_model_layers_25_post_attention_layernorm/max_abs":6.03125,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/std":0.00010908122299552094,"train/train/tensor_act_model_layers_27_mlp/max_abs":0.5859375,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/norm":0.0007082378696052606,"train/train/tensor_act_model_layers_77_mlp_up_proj/max_abs":3.203125,"train/train/tensor_act_model_layers_2_mlp_down_proj/mean":-0.0095672607421875,"train/train/tensor_act_model_layers_34_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/norm":0.022505756939733758,"train/train/tensor_param_model_layers_60_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_60_mlp_down_proj/std":0.1086428829286618,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/mean":-2.93925404548645e-06,"train/train/tensor_act_model_layers_39_self_attn_v_proj/std":0.3242187781584807,"train/train/tensor_act_model_layers_24_self_attn_o_proj/mean":-0.00018164515495300293,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/std":0.00010210775244658035,"train/train/tensor_act_model_layers_52_mlp_up_proj/std":0.3906251525878608,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/mean":-0.00014495849609375,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/norm":0.00342167925383646,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/norm":3.546875,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/std":3.900005208937284e-05,"train/train/tensor_act_model_layers_84_mlp_down_proj/max_abs":2.15625,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/std":4.189229640120937e-05,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/mean":4.472211003303528e-06,"train/train/tensor_act_model_layers_14_post_attention_layernorm/norm":5792.611816407392,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/mean":0.000270843505859375,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_mlp_up_proj/norm":5985.524229351149,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/mean":2.7060508728027344e-05,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/max_abs":0.00093841552734375,"train/train/tensor_act_model_layers_1_mlp/mean":-0.00711822509765625,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/mean":0.0005645751953125,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/mean":3.7997961044311523e-07,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/std":2.358042084924375e-05,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/norm":11.5,"train/train/tensor_act_model_layers_68_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/norm":0.001509241958963593,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/max_abs":0.000507354736328125,"train/train/layer_model_layers_57/grad/std":6.110788663412927e-05,"train/train/tensor_act_model_layers_68_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_56/param/std":0.05292198543434511,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/std":0.052490234375,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/mean":5.242996849119663e-06,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/norm":0.0016500009903511954,"train/train/tensor_param_model_layers_52_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/mean":-0.000640869140625,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/std":3.0287253297930404e-05,"train/train/tensor_act_model_layers_84_mlp/std":0.23291070481201764,"train/train/layer__model_layers_76/param/std":0.05680113233442161,"train/train/tensor_act_model_layers_23_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/std":2.2820030746282705e-05,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/mean":2.5902409106492996e-08,"train/train/layer_model_layers_7/act/frac_near_user_limit":0,"train/train/layer_model_layers_65/act/mean":-0.010542094707489014,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/norm":0.0013566760460144952,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/mean":-2.3655593395233154e-07,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_post_attention_layernorm/std":0.996096165972005,"train/train/tensor_act_model_layers_32_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/mean":1.2271106243133545e-05,"train/train/layer__model_layers_20/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp_down_proj/std":0.18774461730653447,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/max_abs":0.1943359375,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_norm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_mlp_down_proj/std":0.12048520238544266,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/norm":0.028778318887538564,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_18_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_61/act/max_abs":13.6875,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/max_abs":0.00023174285888671875,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/max_abs":0.228515625,"train/train/layer_model_layers_78/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/norm":5.125,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_41_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11/mean":0.001430511474609375,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp_up_proj/max_abs":2.875,"train/train/tensor_act_model_layers_80_self_attn/norm":1381.2628996475278,"train/train/tensor_act_model_layers_30_input_layernorm/norm":5792.608154299031,"train/train/tensor_act_model_layers_75_self_attn_q_proj/std":1.0136776668552172,"train/train/tensor_param_model_layers_73_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/norm":7.8125,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/mean":-0.000209808349609375,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/max_abs":0.00128173828125,"train/train/layer__model_layers_89/param/mean":0.001267514995033395,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_input_layernorm/norm":5792.608642586824,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/std":7.440767711897845e-05,"train/train/tensor_act_model_layers_23_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/norm":4.90625,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_mlp_down_proj/max_abs":0.69140625,"train/train/tensor_act_model_layers_33_self_attn_q_proj/norm":5686.942512725691,"train/train/tensor_act_model_layers_84_self_attn_q_proj/mean":-0.055419921875,"train/train/layer_model_layers_26/act/mean":-0.0035715836745042065,"train/train/tensor_act_model_layers_89_post_attention_layernorm/max_abs":5.34375,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_v_proj/mean":0.00141143798828125,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/mean":0.000514984130859375,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/mean":-0.000568389892578125,"train/train/tensor_act_model_layers_71_self_attn_v_proj/mean":0.00014215707778930664,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/mean":1.4959368854761124e-08,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/norm":4.59375,"train/train/layer_model_layers_75/act/max_abs":12.75,"train/train/tensor_act_model_layers_27_self_attn_o_proj/mean":0.0005831718444824219,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/norm":6.5625,"train/train/tensor_act_model_layers_72_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_33/param/norm":20.257768812621617,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/std":0.0252685546875,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/std":0.00014783192727769247,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/mean":-1.2755393981933594e-05,"train/train/tensor_act_model_layers_66_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/max_abs":0.1591796875,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/norm":0.01251522120944755,"train/train/tensor_act_model_layers_41_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/norm":5.5,"train/train/tensor_act_model_layers_43/norm":7521.924231143352,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/max_abs":0.000644683837890625,"train/train/tensor_act_model_layers_79_self_attn/norm":1005.9306311325272,"train/train/tensor_act_model_layers_29_self_attn_o_proj/std":0.052369167585141096,"train/train/global/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_post_attention_layernorm/max_abs":5.90625,"train/train/tensor_act_model_layers_9_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_input_layernorm/mean":0.071533203125,"train/train/tensor_act_model_layers_23_self_attn_v_proj/std":0.3515625290365671,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/mean":-2.682209014892578e-05,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn/max_abs":0.640625,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/norm":8.875,"train/train/tensor_act_model_layers_24_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_input_layernorm/std":1.0000001415610211,"train/train/tensor_act_model_layers_37_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_mlp_up_proj/std":0.46533310400728095,"train/train/tensor_act_model_layers_15_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn_v_proj/norm":2139.1536796909763,"train/train/tensor_act_model_layers_88_input_layernorm/norm":5792.607666018326,"train/train/tensor_act_model_layers_43_mlp/mean":-0.0018405914306640625,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/norm":0.005595774725011539,"train/train/layer__model_layers_72/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp_up_proj/std":0.5859377034504855,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/norm":4.28125,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/std":0.044677734375,"train/train/tensor_act_model_layers_19_self_attn_k_proj/max_abs":4.5625,"train/train/tensor_act_model_layers_88_self_attn/max_abs":3.828125,"train/train/tensor_act_model_layers_50_mlp_up_proj/std":0.3750001589456857,"train/train/layer__model_layers_92/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/std":2.6841346221383254e-05,"train/train/tensor_act_model_layers_71_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/max_abs":0.0012359619140625,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/mean":1.71661376953125e-05,"train/train/tensor_act_model_layers_23_self_attn_q_proj/norm":6203.23214360874,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/max_abs":0.1416015625,"train/train/tensor_act_model_layers_58_post_attention_layernorm/norm":5792.60449219177,"train/train/tensor_act_model_layers_80_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/norm":0.022492676576237366,"train/train/layer_model_layers_38/act/max_abs":17.25,"train/train/layer__model_layers_57/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/std":1.7324253432244173e-05,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/max_abs":0.189453125,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/norm":0.0006712539498225273,"train/train/tensor_act_model_layers_35_mlp_up_proj/frac_near_user_limit":0,"train/train/layer__model_layers_17/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/mean":1.1856900528073311e-07,"train/train/tensor_act_model_layers_45/max_abs":16.375,"train/train/tensor_param_model_layers_35_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_78_self_attn_o_proj/mean":0.00043952465057373047,"train/train/tensor_act_model_layers_49_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/std":0.0283203125,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/std":4.150302500937909e-05,"train/train/layer__model_layers_20/param/mean":0.0016505439270305187,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_q_proj/norm":5336.086526584793,"train/train/tensor_act_model_layers_90_self_attn_o_proj/mean":-0.0008215904235839844,"train/train/tensor_act_model_layers_2_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_v_proj/mean":-0.004299163818359375,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_24_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/std":0.0291748046875,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp_up_proj/std":0.4941409265570041,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/norm":0.0024428006005128497,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/max_abs":7.772445678710938e-05,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/mean":-2.0712614059448242e-06,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/norm":6.3125,"train/train/tensor_act_model_layers_37_self_attn_k_proj/std":0.9013730371312271,"train/train/tensor_act_model_layers_24/norm":7787.339983230293,"train/train/tensor_act_model_layers_90_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/norm":5.15625,"train/train/tensor_act_model_layers_9_self_attn_v_proj/max_abs":3.078125,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/mean":3.844499588012695e-06,"train/train/tensor_param_model_layers_65_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/mean":0.000152587890625,"train/train/tensor_act_model_layers_70_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/norm":5.875,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/mean":4.169531166553497e-06,"train/train/layer__model_layers_19/param/mean":0.0018014178819105889,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/mean":0.00089263916015625,"train/train/tensor_act_model_layers_22_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/mean":-9.676441550254822e-07,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/norm":5.625,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/mean":-2.5510787963867188e-05,"train/train/layer_model_layers_13/grad/norm":0.032610226228454456,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/mean":-1.8533319234848022e-06,"train/train/tensor_act_model_layers_53_self_attn_q_proj/mean":0.0404052734375,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/std":0.048828125,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/mean":-2.2398307919502258e-07,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/norm":9.25,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/std":5.3114866413479407e-05,"train/train/tensor_act_model_layers_35_self_attn_k_proj/norm":5105.335187996106,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/mean":0.0001621246337890625,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_20_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/norm":5.90625,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_v_proj/max_abs":2.75,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/mean":6.683170795440674e-06,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/mean":1.6689300537109375e-05,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/max_abs":6.723403930664062e-05,"train/train/tensor_act_model_layers_57_post_attention_layernorm/std":1.0000001415610211,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/norm":0.013597113697391505,"train/train/tensor_act_model_layers_77_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/std":5.2757332688618275e-05,"train/train/tensor_act_model_layers_15_self_attn_v_proj/max_abs":2.03125,"train/train/tensor_act_model_layers_35_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_v_proj/std":0.2919938528370652,"train/train/tensor_act_model_layers_46_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_k_proj/mean":-0.0167236328125,"train/train/tensor_act_model_layers_78_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn/norm":1401.4575336815262,"train/train/tensor_act_model_layers_29_mlp_down_proj/norm":424.34578283162824,"train/train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/std":0.036865234375,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp/mean":0.024200439453125,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/max_abs":0.0001316070556640625,"train/train/tensor_act_model_layers_55_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn/max_abs":0.48046875,"train/train/layer_model_layers_85/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/norm":0.028878466682901897,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/max_abs":0.00067138671875,"train/train/tensor_act_model_layers_26_self_attn/mean":-0.000698089599609375,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/max_abs":0.000476837158203125,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/norm":0.0007354999474214281,"train/train/tensor_act_model_layers_60_mlp/max_abs":1.2265625,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/norm":3.15625,"train/train/tensor_act_model_layers_11_self_attn_v_proj/mean":0.003650665283203125,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_27/act/mean":0.0022558799156775842,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/max_abs":0.0002727508544921875,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/max_abs":0.001129150390625,"train/train/tensor_act_model_layers_0_self_attn_v_proj/norm":1328.6417446705814,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/max_abs":0.251953125,"train/train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/std":1.6674677433517956e-05,"train/train/tensor_act_model_layers_42_mlp_up_proj/norm":3396.601464972047,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/norm":0.018771362056617443,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/mean":-0.000339508056640625,"train/train/tensor_act_model_layers_13_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/max_abs":0.00014972686767578125,"train/train/tensor_act_model_layers_54_self_attn_v_proj/norm":1975.4340366374686,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/std":9.951118170440424e-05,"train/train/tensor_act_model_layers_84_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn/mean":-0.00021070241928100586,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_77_self_attn/norm":931.3154026047223,"train/train/tensor_act_model_layers_43_input_layernorm/mean":0.0638427734375,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/std":0.0380859375,"train/train/tensor_act_model_layers_63_self_attn_k_proj/std":0.8857438866579019,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_56/grad/max_abs":0.000682830810546875,"train/train/layer_model_layers_78/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn_o_proj/mean":0.0012607574462890625,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21/std":1.3554969298212052,"train/train/tensor_act_model_layers_9_self_attn_q_proj/std":1.8320355242453619,"train/train/tensor_act_model_layers_77/norm":9275.463944468656,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/mean":0.0001163482666015625,"train/train/tensor_act_model_layers_66_mlp_up_proj/max_abs":3.375,"train/train/tensor_param_model_layers_31_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/norm":0.030916159330474226,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/max_abs":0.1533203125,"train/train/tensor_grad_model_embed_tokens_weight/mean":-2.812594175338745e-07,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/max_abs":0.125,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/norm":0.0318210403130622,"train/train/layer_model_layers_16/act/std":0.6570320160894837,"train/train/tensor_act_model_layers_83_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/std":6.895557705914443e-05,"train/train/layer_model_layers_69/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_self_attn_k_proj/mean":0.03302001953125,"train/train/tensor_act_model_layers_33_mlp_up_proj/norm":2892.3943352249094,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/std":1.71134024189108e-05,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85/std":1.9043054355526736,"train/train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp/norm":8393.684975032415,"train/train/tensor_act_model_layers_87_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/max_abs":0.2001953125,"train/train/tensor_act_model_layers_27_self_attn_v_proj/std":0.3242188479347254,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_k_proj/mean":-0.039306640625,"train/train/tensor_act_model_layers_25_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer__model_layers_58/param/norm":21.659637406432385,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn/norm":335.54180458061114,"train/train/layer_model_layers_19/act/max_abs":17.875,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/mean":-3.93338268622756e-07,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/norm":0.014512127219775444,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/max_abs":0.1552734375,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/mean":-2.256035804748535e-05,"train/train/layer_model_layers_32/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/mean":-2.589295036159456e-07,"train/train/tensor_act_model_layers_61_mlp/max_abs":1.6796875,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_input_layernorm/std":1.0000016912803653,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/max_abs":0.00109100341796875,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/max_abs":0.11865234375,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/mean":-0.0014801025390625,"train/train/layer__model_layers_20/param/max_abs":1,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/std":3.7761358528123056e-05,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/norm":0.0025239693011906094,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/norm":0.029083812343966543,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/std":0.024658203125,"train/train/tensor_act_model_layers_21_post_attention_layernorm/norm":5792.604370117963,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/norm":3.78125,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/std":6.402373351339874e-05,"train/train/tensor_act_model_layers_77_self_attn_v_proj/norm":2615.2760313201184,"train/train/layer_model_layers_71/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_up_proj/norm":7862.927113018212,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_15/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/std":0.0286865234375,"train/train/layer_model_layers_15/grad/max_abs":0.0016326904296875,"train/train/tensor_act_model_layers_48_self_attn_o_proj/std":0.07019165411171444,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/mean":4.172325134277344e-06,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/norm":0.007783971794012197,"train/train/tensor_act_model_layers_48/std":1.300787229781969,"train/train/tensor_act_model_layers_48_self_attn/mean":-0.00118255615234375,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_up_proj/max_abs":3.84375,"train/train/tensor_act_model_layers_74_mlp_down_proj/norm":935.1437962209762,"train/train/tensor_param_model_layers_74_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_43/param/max_abs":1,"train/train/tensor_act_model_layers_60_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_input_layernorm/std":1.0000000562285987,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/max_abs":0.09326171875,"train/train/tensor_act_model_layers_40_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_30/param/mean":0.0015360315950723967,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/norm":0.011494112687251154,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/max_abs":0.00110626220703125,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/mean":1.1252996046096087e-07,"train/train/layer_model_layers_87/act/mean":-0.007129595829890325,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/norm":0.018989880238477287,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp/norm":448.9106466684905,"train/train/layer_model_layers_71/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/max_abs":0.0005950927734375,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/std":2.0409835528127412e-05,"train/train/tensor_act_model_layers_67/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_v_proj/mean":0.0020084381103515625,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/mean":2.0489096641540527e-08,"train/train/layer_model_layers_83/act/max_abs":11.875,"train/train/tensor_act_model_layers_72_mlp/norm":952.6632826005591,"_runtime":3738,"train/train/tensor_param_model_layers_47_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_15_mlp_up_proj/mean":-0.0599365234375,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/std":0.025390625,"train/train/tensor_act_model_layers_19_self_attn_o_proj/std":0.024506068810946684,"train/train/tensor_act_model_layers_7_mlp_up_proj/norm":2428.399618825524,"train/train/tensor_act_model_layers_57_mlp_down_proj/max_abs":1.3046875,"train/train/layer_model_layers_91/act/std":1.0227779314799663,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/max_abs":0.000583648681640625,"train/train/tensor_act_model_layers_16_self_attn_k_proj/max_abs":4.75,"train/train/layer_model_layers_77/act/norm":15838.919834704415,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/norm":0.015454856046606624,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/mean":-0.0004253387451171875,"train/train/tensor_param_model_layers_87_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_31_mlp_down_proj/mean":-0.0007801055908203125,"train/train/tensor_act_model_layers_92_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp/mean":-0.0044708251953125,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/max_abs":0.1416015625,"train/train/tensor_act_model_layers_65_self_attn_q_proj/max_abs":6.34375,"train/train/layer_model_layers_19/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_o_proj/norm":2257.7352672049547,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/mean":0.00012969970703125,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/std":0.051025390625,"train/train/tensor_act_model_layers_90_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/mean":2.7194619178771973e-07,"train/train/tensor_act_model_layers_80_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25/mean":0.0386962890625,"train/train/tensor_act_model_layers_71_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_64/param/mean":0.001362674135871685,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/std":2.5913345384693194e-05,"train/train/layer_model_layers_42/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_66/act/max_abs":13.375,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/max_abs":0.255859375,"train/train/tensor_act_model_layers_11_self_attn/norm":251.28645741320037,"train/train/tensor_act_model_layers_69_self_attn_o_proj/mean":-0.000797271728515625,"train/train/layer_model_layers_65/act/norm":14804.262599067215,"train/train/layer_model_layers_28/act/norm":14316.511600009255,"train/train/tensor_act_model_layers_48_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/std":0.0419921875,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/std":0.00021352843015708334,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/norm":3.125,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/mean":-2.7179718017578125e-05,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/mean":4.291534423828125e-05,"train/train/tensor_act_model_layers_7_self_attn_k_proj/std":1.5761756787614647,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_norm_weight/norm":0.07044504177062214,"train/train/layer__model_layers_45/param/mean":0.001558277052761798,"train/train/layer__model_layers_15/param/norm":19.92237131734699,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/std":6.232587903252136e-05,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn_q_proj/max_abs":5.53125,"train/train/layer__model_layers_79/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/norm":0.046085221099588014,"train/train/tensor_act_model_layers_26_self_attn_v_proj/mean":-0.0009403228759765625,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/norm":4.71875,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/max_abs":0.0001621246337890625,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/std":0.0419921875,"train/train/tensor_act_model_layers_65/norm":8202.354460085198,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_79_mlp_up_proj/norm":5392.40306168714,"train/train/tensor_act_model_layers_29_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/norm":0.03321120219998974,"train/train/tensor_act_model_layers_27_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/norm":0.020981273299753425,"train/train/tensor_act_model_layers_79_self_attn_k_proj/std":0.9335939415328974,"train/train/tensor_act_model_layers_33_self_attn_q_proj/mean":-0.1141357421875,"train/train/tensor_act_model_layers_84_self_attn_q_proj/norm":6197.741888683267,"train/train/tensor_act_model_layers_22_self_attn_k_proj/max_abs":4.90625,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/max_abs":0.000278472900390625,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/mean":6.183981895446777e-07,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/norm":0.02525004570550842,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/std":0.0242919921875,"train/train/tensor_act_model_layers_46_self_attn_o_proj/max_abs":1.375,"train/train/tensor_act_model_layers_4/max_abs":18.5,"train/train/tensor_act_model_layers_15_self_attn/norm":264.3069993316272,"train/train/layer_model_layers_20/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_v_proj/max_abs":2.09375,"train/train/layer_model_layers_4/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_k_proj/norm":4217.759489373933,"train/train/layer__model_layers_90/param/std":0.06298860499053334,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/max_abs":0.00018024444580078125,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/norm":0.020845538776056858,"train/train/tensor_act_model_layers_0_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/max_abs":0.25390625,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/norm":6.78125,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/norm":4.375,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/std":7.914795171412513e-05,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/max_abs":0.00189208984375,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/max_abs":0.000255584716796875,"train/train/tensor_param_model_layers_54_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/mean":-0.0003223419189453125,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/std":0.00010327216952556936,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/std":6.939177804456352e-05,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/max_abs":0.0003452301025390625,"train/train/tensor_act_model_layers_68_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp/max_abs":0.59375,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/max_abs":0.002105712890625,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/std":0.034423828125,"train/train/tensor_act_model_layers_8_input_layernorm/max_abs":5.46875,"train/train/tensor_act_model_layers_63_self_attn_q_proj/max_abs":7.375,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/mean":-7.718335837125778e-08,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_90/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn/std":0.09069860037621252,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_5_self_attn_o_proj/norm":427.94843800453395,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/norm":0.02305311668057658,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/max_abs":0.00119781494140625,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/max_abs":0.232421875,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/norm":4.84375,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/mean":1.8495484255254269e-07,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/mean":-0.00018024444580078125,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/std":4.554607019931789e-05,"train/train/tensor_act_model_layers_37_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_k_proj/mean":-0.018798828125,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/std":8.239540637514342e-05,"train/train/tensor_act_model_layers_76_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/mean":0.00013637542724609375,"train/train/tensor_act_model_layers_86_self_attn_q_proj/max_abs":5.59375,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/max_abs":0.00110626220703125,"train/train/tensor_act_model_layers_1_self_attn_q_proj/std":1.3750000650232472,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/norm":6.28125,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn_q_proj/mean":-0.0465087890625,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/max_abs":0.1201171875,"train/train/tensor_act_model_layers_69_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer__model_layers_70/param/max_abs":1,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/std":0.0284423828125,"train/train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/max_abs":0.00060272216796875,"train/train/tensor_act_model_layers_22_self_attn_o_proj/norm":364.0287033922374,"train/train/layer_model_layers_2/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer__model_layers_50/param/mean":0.0014729135307990444,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/std":0.00013568121097487667,"train/train/layer__model_layers_11/param/mean":0.001915918311537149,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/std":0.00015936276547954997,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/max_abs":0.13671875,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/max_abs":0.000885009765625,"train/train/tensor_act_model_layers_31_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/mean":4.2421743273735046e-07,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/norm":0.0007606950171100162,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/max_abs":0.000339508056640625,"train/train/tensor_act_model_layers_46_input_layernorm/std":0.9960938322777808,"train/train/layer_model_layers_64/grad/frac_near_user_limit":0,"train/train/layer_model_layers_47/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/norm":0.0010852344415637671,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/max_abs":0.125,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/norm":0.0006919681037829634,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/norm":7.78125,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_k_proj/std":0.7949243868851399,"train/train/tensor_act_model_layers_54_input_layernorm/std":1.0000001993030112,"train/train/tensor_act_model_layers_43_self_attn_q_proj/std":0.9882812993564141,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/max_abs":0.00013446807861328125,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/norm":0.01285387928495649,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/mean":-0.0003604888916015625,"train/train/layer_model_layers_58/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/norm":0.03739219109411383,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/std":0.00012309668849842855,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/std":0.0235595703125,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_post_attention_layernorm/mean":0.047607421875,"train/train/tensor_act_model_layers_39_self_attn_q_proj/mean":-0.0014461874961853027,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/mean":-4.9591064453125e-05,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/std":3.7958381500108574e-05,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/norm":4,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/std":3.7207275383938666e-05,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/max_abs":0.1337890625,"train/train/tensor_act_model_layers_39_self_attn_q_proj/max_abs":6.34375,"train/train/tensor_act_model_layers_32_mlp_down_proj/norm":264.5384415227059,"train/train/tensor_act_model_layers_54_mlp_up_proj/mean":-0.0953369140625,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/norm":0.01663821046173546,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/mean":-0.000579833984375,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/norm":0.02082560316359488,"train/train/tensor_act_model_layers_67_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/mean":3.810971975326538e-06,"train/train/tensor_act_model_layers_64_input_layernorm/norm":5792.605346681482,"train/train/layer_model_layers_24/act/max_abs":17.75,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8/std":1.46096885306137,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/std":7.910273068811586e-05,"train/train/tensor_act_model_layers_46_mlp/max_abs":0.9140625,"train/train/tensor_act_model_layers_63_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/max_abs":0.2099609375,"train/train/tensor_act_model_layers_63_mlp_down_proj/max_abs":1.03125,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/max_abs":0.2236328125,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_k_proj/max_abs":4.625,"train/train/tensor_act_model_layers_78_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_26/grad/frac_near_user_limit":0,"train/train/layer_model_layers_49/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_k_proj/max_abs":5.34375,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/mean":3.552436828613281e-05,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/norm":0.017616336404410814,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/max_abs":0.0010528564453125,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/mean":-0.00024127960205078125,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_5/act/norm":16439.667220347263,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/norm":0.038697251750150506,"train/train/tensor_act_model_layers_68/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/std":0.023681640625,"train/train/tensor_act_model_layers_13_post_attention_layernorm/std":1.0000000512641225,"train/train/tensor_act_model_layers_12/mean":0.00348663330078125,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/max_abs":0.1181640625,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/mean":6.522077455883846e-08,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/max_abs":0.0006103515625,"train/train/tensor_param_model_layers_30_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/max_abs":0.107421875,"train/train/tensor_act_model_layers_37_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/mean":0.0001049041748046875,"train/train/tensor_act_model_layers_42_self_attn_k_proj/mean":-0.00470733642578125,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/max_abs":0.000919342041015625,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/mean":3.3527612686157227e-06,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_15_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_post_attention_layernorm/max_abs":5.6875,"train/train/tensor_act_model_layers_77_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/max_abs":0.1396484375,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/norm":4.03125,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/std":0.0230712890625,"train/train/tensor_act_model_layers_37_mlp_down_proj/norm":341.4315058516043,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/mean":6.984919309616089e-07,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/mean":-1.2191012501716614e-06,"train/train/tensor_act_model_layers_59/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/std":0.05517578125,"train/train/tensor_act_model_layers_3_mlp_down_proj/max_abs":0.671875,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/mean":1.9744038581848145e-06,"train/train/tensor_act_model_layers_24_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_o_proj/max_abs":1.4296875,"train/train/layer__model_layers_55/param/std":0.05366909064296834,"train/train/tensor_act_model_layers_36_post_attention_layernorm/max_abs":6.03125,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/mean":1.2740492820739746e-06,"train/train/tensor_act_model_layers_23_self_attn/max_abs":1.5546875,"train/train/layer__model_layers_82/param/max_abs":1,"train/train/tensor_act_model_layers_93_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/mean":2.0558945834636688e-07,"train/train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88/mean":0.157958984375,"train/train/tensor_act_model_layers_63_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_50_self_attn_o_proj/mean":-0.001644134521484375,"train/train/layer_model_layers_70/grad/mean":1.2522845205018376e-06,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_32/grad/max_abs":0.00136566162109375,"train/train/tensor_act_model_layers_36_mlp_up_proj/norm":3124.4997237124862,"train/train/tensor_act_model_layers_86_self_attn_o_proj/norm":1016.720683611927,"train/train/layer_model_layers_64/act/max_abs":13.4375,"train/train/tensor_act_model_layers_46_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_o_proj/std":0.12598226066108384,"train/train/tensor_act_model_layers_84/max_abs":11.5,"train/train/tensor_act_model_layers_17_self_attn/std":0.04382387288809894,"train/train/tensor_act_model_layers_5_post_attention_layernorm/norm":5792.614257816955,"train/train/tensor_act_model_layers_10_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/norm":0.012201269697441723,"train/train/tensor_act_model_layers_66_self_attn_o_proj/max_abs":2.671875,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn/norm":541.5968840862648,"train/train/tensor_act_model_layers_66_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18/norm":7930.962874787707,"train/train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_o_proj/norm":719.1592407583274,"train/train/tensor_act_model_layers_86_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_79/grad/frac_near_user_limit":0,"train/train/layer_model_layers_85/act/max_abs":11.25,"train/train/tensor_act_model_layers_4_mlp/std":0.06591796875,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/norm":0.06208004528871094,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/max_abs":0.000896453857421875,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/std":9.826061829292108e-05,"train/train/layer_model_layers_56/grad/mean":-1.3130470527722477e-06,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_post_attention_layernorm/std":1.0000014454115913,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/max_abs":0.25,"train/train/layer_model_layers_87/act/frac_near_user_limit":0,"train/train/layer__model_layers_63/param/max_abs":1,"train/train/tensor_act_model_layers_13_input_layernorm/max_abs":5.9375,"train/train/tensor_act_model_layers_47_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/norm":7,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/mean":-0.00037384033203125,"train/train/tensor_act_model_layers_48_mlp_down_proj/norm":480.44002816296006,"train/train/tensor_act_model_layers_10_post_attention_layernorm/norm":5792.60546875105,"train/train/tensor_act_model_layers_80_self_attn/mean":-0.003803253173828125,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/norm":0.06592429801789954,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/std":9.24316248507987e-06,"train/train/tensor_act_model_layers_46/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_32/grad/norm":0.030488562200888022,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_k_proj/max_abs":4.25,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/norm":0.0034571008351968286,"train/train/layer__model_layers_2/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_q_proj/max_abs":5.71875,"train/train/tensor_act_model_layers_38/norm":7576.6104812976255,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_78/act/norm":15930.709290073633,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/mean":6.437301635742188e-05,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_down_proj/mean":0.001171112060546875,"train/train/tensor_act_model_layers_48_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn_o_proj/mean":-0.00699615478515625,"train/train/tensor_param_model_layers_92_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_19_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/norm":0.019398716439474294,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/std":0.0341796875,"train/train/tensor_act_model_layers_90_self_attn_q_proj/std":0.9433654938967322,"train/train/tensor_act_model_layers_82_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp/norm":232.67016588797628,"train/train/tensor_act_model_layers_66/std":1.4355519028659927,"train/train/tensor_act_model_layers_51_self_attn/std":0.07080417966085727,"train/train/layer__model_layers_16/param/std":0.04777227771287646,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/std":1.0174891664090927e-05,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_18/act/std":0.6676630397234755,"train/train/tensor_act_model_layers_63_self_attn_o_proj/max_abs":1.8125,"train/train/tensor_act_model_layers_82_mlp_up_proj/std":0.585937550862628,"train/train/tensor_act_model_layers_52_self_attn_v_proj/mean":0.00025653839111328125,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/max_abs":0.000667572021484375,"train/train/tensor_act_model_layers_12_self_attn/mean":-0.0015048980712890625,"train/train/tensor_act_model_layers_77_mlp_up_proj/norm":5324.541603193532,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/mean":-1.6726553440093994e-06,"train/train/tensor_act_model_layers_11_mlp_up_proj/std":0.20800791995623324,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/max_abs":0.0003299713134765625,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/mean":-4.854518920183182e-08,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/norm":0.0026223373277130153,"train/train/tensor_act_model_layers_53_self_attn/max_abs":1.078125,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/max_abs":0.1162109375,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/std":4.289949848109671e-05,"train/train/tensor_act_model_layers_10_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_23/act/norm":14348.120248448544,"train/train/tensor_act_model_layers_87_self_attn_v_proj/mean":0.008819580078125,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/mean":0.0002613067626953125,"train/train/layer__model_layers_31/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_norm/max_abs":5.25,"train/train/layer_model_layers_56/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/std":2.7104195366459027e-05,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/norm":0.013431409852970234,"train/train/tensor_act_model_layers_80_self_attn_v_proj/std":0.4843750521003132,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/mean":-2.0209699869155884e-06,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/std":1.1138018996278328e-05,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/norm":6.28125,"train/train/tensor_act_model_layers_9_mlp_down_proj/norm":269.68675804093186,"train/train/tensor_act_model_layers_13_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/max_abs":0.11376953125,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_90/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/norm":3.84375,"train/train/layer__model_layers_74/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32/max_abs":17.75,"train/train/tensor_act_model_layers_30_self_attn/mean":-0.00384521484375,"train/train/tensor_act_model_layers_87_mlp_down_proj/max_abs":4.15625,"train/train/tensor_act_model_layers_86_self_attn_o_proj/max_abs":2.71875,"train/train/layer__model_layers_17/param/max_abs":1,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_k_proj/std":1.1015627130548797,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/mean":-5.1961251301690936e-08,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_o_proj/std":0.17529367442388719,"train/train/tensor_act_model_layers_14_post_attention_layernorm/std":1.0000000410000203,"train/train/tensor_act_model_layers_9_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn/std":0.07385304414240479,"train/train/tensor_act_model_layers_72_self_attn_q_proj/mean":0.06219482421875,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_up_proj/mean":-0.05157470703125,"train/train/tensor_act_model_layers_72_mlp_down_proj/max_abs":1.3203125,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_3/param/std":0.048769678800471124,"train/train/layer_model_layers_3/act/max_abs":18.5,"train/train/tensor_act_model_layers_0_self_attn_o_proj/mean":0.001895904541015625,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/mean":-1.4994293451309204e-07,"train/train/layer_model_layers_23/grad/std":5.16168775729547e-05,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/mean":-9.262561798095703e-05,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/norm":0.037390198491139184,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_37_self_attn_v_proj/norm":1912.4447411557464,"train/train/tensor_act_model_layers_39_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_input_layernorm/norm":5792.609619149871,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/norm":5.1875,"train/train/tensor_act_model_layers_11_mlp/mean":0.00628662109375,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/std":9.824496030537473e-05,"train/train/layer_model_layers_53/grad/mean":-1.3844647652552392e-06,"train/train/tensor_act_model_layers_39_mlp_down_proj/max_abs":0.671875,"train/train/layer_model_layers_62/grad/max_abs":0.001312255859375,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/std":0.0311279296875,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/norm":5.5,"train/train/tensor_act_model_layers_70_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/std":0.033447265625,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/mean":-6.914138793945312e-05,"train/train/tensor_act_model_layers_89_self_attn_q_proj/mean":0.117431640625,"train/train/tensor_act_model_layers_75_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_k_proj/max_abs":5.0625,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/mean":9.255018085241318e-08,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/norm":6.28125,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/std":0.00012225596170182602,"train/train/tensor_act_model_layers_48_self_attn_k_proj/max_abs":5.125,"train/train/tensor_act_model_layers_7_post_attention_layernorm/std":1.0000000119616743,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/std":1.2382446815372734e-05,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn_o_proj/mean":-0.00037097930908203125,"train/train/tensor_act_model_layers_29_self_attn/std":0.052369167585141096,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp/norm":916.2282459230611,"train/train/tensor_act_model_layers_2_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_k_proj/mean":-0.0116729736328125,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/max_abs":0.000247955322265625,"train/train/tensor_act_model_layers_39_self_attn_o_proj/max_abs":1.046875,"train/train/layer_model_layers_33/grad/mean":-6.796330120559787e-07,"train/train/layer_model_layers_4/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/std":8.269760128233022e-06,"train/train/tensor_act_model_layers_10_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/norm":0.01617765572102903,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/mean":-1.5425030142068863e-08,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_input_layernorm/mean":0.0694580078125,"train/train/layer_model_layers_53/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_o_proj/mean":-0.003017425537109375,"train/train/tensor_act_model_layers_28_mlp/norm":284.54347916169047,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/mean":8.026836439967155e-08,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/norm":0.0045554833003220475,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/std":6.056251785981836e-05,"train/train/tensor_act_model_layers_4_self_attn_v_proj/norm":1407.6614908935226,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/std":0.00010344805312786797,"train/train/tensor_param_model_layers_64_input_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_62/param/max_abs":1,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/std":0.025634765625,"train/train/tensor_act_model_layers_28_self_attn_v_proj/mean":-0.00357818603515625,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/mean":1.4398574421647936e-07,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/norm":0.014182730471710125,"train/train/layer__model_layers_34/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_16_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/max_abs":0.1083984375,"train/train/tensor_act_model_layers_53_self_attn_q_proj/norm":5202.730941202912,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1/norm":8661.686518824228,"train/train/layer__model_layers_36/param/mean":0.001448799407054407,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/std":0.00012973138454140106,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/std":0.049560546875,"train/train/tensor_act_model_layers_69_mlp/std":0.16748143910385493,"train/train/layer__model_layers_2/param/max_abs":1,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/mean":-5.435943603515625e-05,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/norm":7.6875,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_post_attention_layernorm/max_abs":5.375,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn_q_proj/norm":10632.903596887367,"train/train/tensor_act_model_layers_72_mlp/mean":0.016204833984375,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/max_abs":0.000560760498046875,"train/train/tensor_act_model_layers_35_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/mean":-1.9166618585586548e-06,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/mean":6.341934204101562e-05,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_71_mlp_up_proj/mean":-0.0933837890625,"train/train/tensor_act_model_layers_13_mlp/mean":0.003505706787109375,"train/train/tensor_act_model_layers_20_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_input_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_65/param/std":0.055859497771026057,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/mean":-0.00020885467529296875,"train/train/tensor_act_model_layers_56_input_layernorm/norm":5792.6104736386915,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/norm":6.8125,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/std":5.910106676821726e-05,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/mean":-9.458744898438454e-09,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp_up_proj/mean":-0.059326171875,"train/train/tensor_act_model_layers_38/max_abs":17.25,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/max_abs":0.00045013427734375,"train/train/tensor_act_model_layers_88_self_attn_q_proj/norm":7056.24551907273,"train/train/tensor_act_model_layers_83_post_attention_layernorm/norm":5792.609130860204,"train/train/tensor_act_model_layers_73_mlp_down_proj/mean":0.010162353515625,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/norm":0.0013609704367996663,"train/train/tensor_act_model_layers_59_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp_down_proj/mean":0.0589599609375,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/max_abs":0.0006561279296875,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/mean":-3.327149897813797e-07,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/norm":11.1875,"train/train/tensor_act_model_layers_93_mlp/mean":0.0589599609375,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/norm":0.0006925829507675232,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/max_abs":0.0003204345703125,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/std":0.0458984375,"train/train/tensor_act_model_layers_53_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/norm":0.020816657226790443,"train/train/tensor_act_model_layers_30_self_attn_v_proj/max_abs":2.40625,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/norm":0.0017150532729161064,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/norm":0.02796249596770952,"train/train/tensor_act_model_layers_67_mlp_down_proj/std":0.13647701721080444,"train/train/layer_model_layers_93/act/norm":27974.74538620114,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/mean":3.8463622331619263e-07,"train/train/layer_model_layers_57/act/std":0.6814372350383208,"train/train/layer_model_layers_35/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/std":0.042724609375,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/max_abs":0.0006866455078125,"train/train/tensor_act_model_layers_73_mlp_down_proj/std":0.16455338756183835,"train/train/tensor_act_model_layers_44_self_attn_v_proj/std":0.402343768373276,"train/train/tensor_act_model_layers_45_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_mlp/max_abs":0.8359375,"train/train/layer_model_layers_20/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/std":0.0556640625,"train/train/layer_model_layers_46/grad/mean":-4.998503424308601e-07,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_36_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/norm":5.21875,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/norm":0.020726356427677672,"train/train/tensor_act_model_layers_19_mlp/max_abs":0.458984375,"train/train/tensor_param_model_layers_48_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_v_proj/norm":2358.084176958478,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/std":1.2988328689832742,"train/train/layer_model_layers_55/grad/max_abs":0.00144195556640625,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/norm":0.0003787684163800689,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp/norm":272.5804017256328,"train/train/tensor_act_model_layers_31_post_attention_layernorm/mean":0.0595703125,"train/train/layer_model_layers_61/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_q_proj/std":1.0625000192838556,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/mean":-2.183951437473297e-06,"train/train/layer_model_layers_27/grad/norm":0.035602546180154175,"train/train/tensor_act_model_layers_65_self_attn_v_proj/std":0.4067392333212118,"train/train/layer_model_layers_53/grad/std":5.152250292020753e-05,"train/train/tensor_act_model_layers_73_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_61/grad/mean":-1.3288273221802972e-07,"train/train/layer_model_layers_65/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_v_proj/max_abs":3.328125,"train/train/tensor_act_model_layers_71_post_attention_layernorm/std":1.0000020153800206,"train/train/tensor_act_model_layers_2_self_attn/max_abs":0.765625,"train/train/layer_model_layers_67/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/norm":7.53125,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/mean":1,"train/train/layer__model_layers_21/param/max_abs":1,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp_down_proj/std":0.22534224736425928,"train/train/tensor_act_model_layers_59_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45/mean":0.0819091796875,"train/train/tensor_act_model_layers_84_self_attn_o_proj/norm":936.057846637477,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_k_proj/mean":0.0160369873046875,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/max_abs":0.000827789306640625,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/mean":1.3031065464019775e-05,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/mean":-0.000507354736328125,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/mean":-2.046581357717514e-07,"train/train/tensor_act_model_layers_49_self_attn_k_proj/norm":4605.885360262698,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_28/act/mean":-0.0038581261268028845,"train/train/tensor_act_model_layers_9_input_layernorm/std":1.0000000156869644,"train/train/tensor_act_model_layers_5_mlp/mean":-0.002838134765625,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp_down_proj/norm":1820.7761108830575,"train/train/tensor_act_model_layers_0/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/norm":0.003515321665499351,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/mean":4.220055416226387e-09,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/std":1.9893139324603955e-05,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_v_proj/norm":1580.1552119270104,"train/train/layer_model_layers_84/act/frac_near_dtype_limit":0,"train/train/layer__model_layers_31/param/mean":0.0016085167942850526,"train/train/tensor_act_model_layers_22_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_post_attention_layernorm/std":1.0000001098960578,"train/train/tensor_act_model_layers_48/frac_near_user_limit":0,"train/train/layer__model_layers_60/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9/mean":-0.0074005126953125,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/norm":0.017281035023571385,"train/train/layer__model_layers_0/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_k_proj/std":0.961915660024178,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/std":7.583740711736887e-05,"train/train/tensor_act_model_layers_13_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_q_proj/norm":5914.462992230686,"train/train/tensor_act_model_layers_15_mlp_down_proj/norm":258.7885784510314,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/std":0.00010124462108647196,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/mean":6.236135959625244e-06,"train/train/tensor_act_model_layers_35_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/std":0.046142578125,"train/train/layer_model_layers_92/act/norm":22717.34673530795,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/std":3.94210548958795e-05,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/mean":-0.00060272216796875,"train/train/layer__model_layers_84/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/max_abs":0.0002155303955078125,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/max_abs":0.00021648406982421875,"train/train/tensor_act_model_layers_74_self_attn_q_proj/std":0.9472697680667844,"train/train/tensor_act_model_layers_72_mlp_down_proj/std":0.1635766185755264,"train/train/tensor_act_model_layers_88_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/std":0.0284423828125,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/norm":7.15625,"train/train/tensor_act_model_layers_6_self_attn_k_proj/std":1.0156251306717128,"train/train/layer__model_layers_76/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_k_proj/mean":-0.024810791015625,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/norm":3.1875,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/norm":0.014714853402689783,"train/train/tensor_act_model_layers_1/mean":0.0345458984375,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/mean":-5.356268957257271e-07,"train/train/tensor_act_model_layers_50_post_attention_layernorm/std":1.000000104308123,"train/train/layer__model_layers_44/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn_o_proj/mean":-0.0012493133544921875,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/max_abs":0.255859375,"train/train/tensor_act_model_layers_20_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn/max_abs":0.32421875,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/max_abs":0.00135040283203125,"train/train/tensor_act_model_layers_2_self_attn/mean":-0.00067901611328125,"train/train/tensor_act_model_layers_62_self_attn/mean":-0.0012493133544921875,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/max_abs":0.126953125,"train/train/tensor_act_model_layers_32/mean":0.0736083984375,"train/train/layer__model_layers_86/param/max_abs":1,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_33/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_q_proj/max_abs":5.3125,"train/train/tensor_act_model_layers_71_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/norm":6.71875,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/std":0.0341796875,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/mean":1.2526288628578186e-07,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_34_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_mlp_down_proj/std":0.16748143910385493,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_87/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn_q_proj/max_abs":7.65625,"train/train/tensor_act_model_layers_44_input_layernorm/mean":0.06195068359375,"train/train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_84/param/std":0.05886651912748472,"train/train/layer_model_layers_31/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/norm":13.4375,"train/train/layer_model_layers_59/grad/mean":-2.313615585707837e-07,"train/train/tensor_param_model_layers_64_input_layernorm_weight/std":0,"train/train/layer_model_layers_10/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/norm":3.234375,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_input_layernorm/std":1.000002518293071,"train/train/tensor_act_model_layers_14_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/mean":-1.3361568562686443e-07,"train/train/tensor_act_model_layers_8_mlp_down_proj/max_abs":0.341796875,"train/train/tensor_act_model_layers_32_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/std":0.051025390625,"train/train/tensor_act_model_layers_0_mlp_up_proj/norm":9651.9206375512,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp_down_proj/norm":213.3784408100035,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/std":0.023681640625,"train/train/tensor_act_model_layers_34_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/norm":0.0017558467655556557,"train/train/tensor_act_model_layers_39_mlp_down_proj/std":0.053161730585522024,"train/train/layer_model_layers_1/grad/norm":0.07027759084841109,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/std":4.04242084667215e-05,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn/norm":557.7745283907466,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_89/grad/mean":3.5367810200789416e-07,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/max_abs":0.2392578125,"train/train/tensor_param_model_layers_24_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/mean":0.0003509521484375,"train/train/layer_model_layers_24/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_k_proj/std":0.8652368307616988,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/std":0.0218505859375,"train/train/tensor_act_model_layers_57_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_post_attention_layernorm/mean":0.025604248046875,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_19/param/max_abs":1,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/std":0.0517578125,"train/train/tensor_act_model_layers_37_input_layernorm/mean":0.0799560546875,"train/train/tensor_act_model_layers_7_self_attn_o_proj/mean":0.0003714561462402344,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/mean":-6.360001862049103e-06,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/max_abs":0.00115966796875,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_q_proj/norm":6171.858606266196,"train/train/tensor_act_model_layers_37/std":1.3066569271695778,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/std":0.0262451171875,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/max_abs":0.000732421875,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/norm":0.0008796145148378683,"train/train/tensor_act_model_layers_48_input_layernorm/std":0.9960938322777808,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/std":6.545020278322656e-05,"train/train/tensor_param_model_layers_84_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/mean":2.410524757578969e-08,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/max_abs":0.166015625,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/mean":-3.033783286809921e-07,"train/train/layer_model_layers_23/grad/norm":0.041803902882589485,"train/train/tensor_param_model_layers_45_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_65_mlp_down_proj/mean":-0.002948760986328125,"train/train/tensor_act_model_layers_6_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_91_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/mean":-0.01123046875,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/norm":0.00550681642899945,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/mean":2.8461217880249023e-06,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/norm":0.0013496429115550605,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_89_mlp_down_proj/mean":-0.042724609375,"train/train/tensor_act_model_layers_36_self_attn_v_proj/norm":2157.509826336855,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_35_mlp_down_proj/norm":285.1341781058447,"train/train/tensor_act_model_layers_66_post_attention_layernorm/norm":5792.612426760883,"train/train/layer__model_layers_31/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp_up_proj/max_abs":3.265625,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/norm":3.71875,"train/train/tensor_act_model_layers_89_self_attn_o_proj/std":0.30763140751502616,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/std":0.048095703125,"train/train/tensor_act_model_layers_7_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/mean":4.3655745685100555e-08,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/norm":3.5625,"train/train/tensor_act_model_layers_38_self_attn_q_proj/norm":5596.271840675148,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/mean":2.0372681319713593e-09,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/norm":0.3546602885693378,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/norm":6.125,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/max_abs":0.0001373291015625,"train/train/tensor_act_model_layers_83_self_attn/norm":1067.5138976407538,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_16_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_29_post_attention_layernorm/std":1.0000000949948982,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/max_abs":0.185546875,"train/train/tensor_act_model_layers_15_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_10/act/max_abs":18.375,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/std":7.618336200070366e-05,"train/train/tensor_act_model_layers_22_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_37/grad/std":4.384741758936541e-05,"train/train/layer_model_layers_28/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84/std":1.84180574222829,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/max_abs":0.0001010894775390625,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_80/grad/mean":1.2121468422100435e-06,"train/train/tensor_act_model_layers_67_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/norm":0.01494666736202546,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40/std":1.3027509605347074,"train/train/tensor_act_model_layers_74_post_attention_layernorm/mean":0.04351806640625,"train/train/tensor_act_model_layers_1_post_attention_layernorm/std":1.0000000223517413,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/max_abs":0.13671875,"train/train/tensor_act_model_layers_39_self_attn/max_abs":1.046875,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/std":0.00011459526320373281,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/std":3.364085302140898e-05,"train/train/layer_model_layers_49/act/std":0.6697020934181188,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/norm":0.017314739020537373,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/max_abs":0.126953125,"train/train/tensor_act_model_layers_41_mlp/max_abs":2.28125,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/mean":-8.061528205871582e-06,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_55/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_18/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn/max_abs":1.140625,"train/train/layer_model_layers_72/act/mean":0.0026672803438626803,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/max_abs":0.142578125,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/max_abs":0.171875,"train/train/tensor_act_model_layers_67_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/std":5.060501938610484e-05,"train/train/tensor_param_model_layers_88_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_70_self_attn_v_proj/std":0.5947291793730435,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_86_mlp/mean":0.0114288330078125,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/mean":-0.00140380859375,"train/train/tensor_act_model_layers_45_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63/max_abs":13.4375,"train/train/tensor_param_model_layers_76_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/mean":0.0002765655517578125,"train/train/tensor_act_model_layers_23/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_1/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_input_layernorm/mean":0.03277587890625,"train/train/tensor_act_model_layers_57_self_attn_v_proj/max_abs":2.875,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/norm":0.02357822134555676,"train/train/tensor_act_model_layers_33_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_73/param/norm":22.520758653129338,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/max_abs":0.15234375,"train/train/tensor_act_model_layers_39_mlp/max_abs":0.671875,"train/train/tensor_act_model_layers_2_input_layernorm/std":1.0000000125728545,"train/train/tensor_act_model_layers_80_self_attn_k_proj/mean":0.10009765625,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn/max_abs":3.5625,"train/train/tensor_param_model_layers_8_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_93_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_mlp_up_proj/mean":-0.103271484375,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/max_abs":0.000576019287109375,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/mean":-0.000423431396484375,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/max_abs":0.00020313262939453125,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/max_abs":0.11962890625,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/mean":-3.440072759985924e-08,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/std":0.05224609375,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_11/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn/norm":275.26248700016043,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/std":3.7620939389591385e-05,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/std":0.039306640625,"train/train/tensor_act_model_layers_62_post_attention_layernorm/max_abs":5.875,"train/train/layer_model_layers_26/grad/max_abs":0.00118255615234375,"train/train/tensor_act_model_layers_46_self_attn_v_proj/mean":0.00567626953125,"train/train/tensor_act_model_layers_80_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/mean":-0.0003833770751953125,"train/train/tensor_act_model_layers_20_input_layernorm/std":1.0000000386498862,"train/train/tensor_act_model_layers_65_self_attn/norm":725.704334990722,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/mean":-0.000301361083984375,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/std":4.9145879090105836e-05,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/mean":-0.000331878662109375,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/std":4.581678236172404e-05,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_15/grad/std":4.221650508007202e-05,"train/train/tensor_act_model_layers_33_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_mlp/norm":362.6507540354738,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/mean":2.7001078706234694e-08,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/max_abs":0.0002536773681640625,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/norm":9.125,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp_down_proj/norm":3525.8202406809833,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/norm":9.4375,"train/train/layer__model_layers_74/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/mean":-5.960464477539063e-08,"train/train/tensor_act_model_layers_48/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/std":3.51774765938552e-05,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/max_abs":0.00049591064453125,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_75/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp_down_proj/max_abs":2.90625,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_84/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/std":6.450345589853723e-05,"train/train/tensor_act_model_layers_30_self_attn/max_abs":1.5234375,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_post_attention_layernorm/max_abs":5.25,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_50/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_13/act/std":0.6924575337223303,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_o_proj/norm":484.48595931229715,"train/train/tensor_act_model_layers_17_self_attn_o_proj/max_abs":0.349609375,"train/train/layer_model_layers_40/act/std":0.6939038568078016,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/std":0.02978515625,"train/train/tensor_act_model_layers_62_self_attn_o_proj/max_abs":1.7109375,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/mean":-1.0370276868343353e-06,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/norm":3.125,"train/train/tensor_act_model_layers_74_input_layernorm/norm":5792.608764651693,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/norm":0.0006380416165668012,"train/train/tensor_act_model_layers_13/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_post_attention_layernorm/max_abs":4.84375,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/norm":0.04603038246803081,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/std":0.040771484375,"train/train/tensor_act_model_layers_49_input_layernorm/norm":5792.609497074607,"train/train/tensor_act_model_layers_26_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/max_abs":0.208984375,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/max_abs":0.0004367828369140625,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/std":8.59550210241344e-05,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/norm":8.1875,"train/train/tensor_act_model_layers_22_mlp_down_proj/norm":256.49137022046824,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/norm":0.002991559437057137,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/norm":0.012807897054949958,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/norm":0.03050811620895687,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/mean":0.00019168853759765625,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/std":1.7804737661440904e-05,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_post_attention_layernorm/max_abs":5.53125,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/norm":0.005354508890602154,"train/train/tensor_act_model_layers_92_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/mean":-0.00020122528076171875,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/std":4.697687415895064e-05,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/mean":-8.0108642578125e-05,"train/train/tensor_act_model_layers_81_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_mlp/std":0.37988416621690385,"train/train/tensor_act_model_layers_18_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_up_proj/max_abs":3.234375,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/mean":-6.079673767089844e-06,"train/train/tensor_act_model_layers_51_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/std":1.1857735706869446e-05,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/norm":2.453125,"train/train/tensor_act_model_layers_27/norm":7694.5364166411055,"train/train/tensor_act_model_layers_46_self_attn_k_proj/norm":5159.720449294364,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/norm":0.0005066946224049318,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/max_abs":0.00067138671875,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/mean":-0.0002651214599609375,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/norm":0.0009735440753742871,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/max_abs":0.1376953125,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/std":0.00010164924944689976,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/max_abs":0.000957489013671875,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/mean":-5.68293035030365e-06,"train/train/tensor_act_model_layers_66_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp/max_abs":0.56640625,"train/train/tensor_act_model_layers_23_self_attn_k_proj/max_abs":5.375,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/norm":6.875,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/norm":7.21875,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/std":2.9499816524723486e-05,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/max_abs":9.870529174804688e-05,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/max_abs":0.1279296875,"train/train/layer_model_layers_42/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/max_abs":0.00148773193359375,"train/train/tensor_act_model_layers_34_post_attention_layernorm/norm":5792.61181641371,"train/train/tensor_act_model_layers_57_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_q_proj/max_abs":7.03125,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/std":9.674557302127334e-05,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/norm":3.890625,"train/train/tensor_act_model_layers_84_self_attn/mean":-0.002216339111328125,"train/train/layer_model_layers_81/grad/std":9.776966356877441e-05,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/mean":7.128715515136719e-05,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/max_abs":0.00061798095703125,"train/train/tensor_act_model_layers_64/mean":0.0623779296875,"train/train/tensor_act_model_layers_86_mlp_up_proj/norm":6253.196349112304,"train/train/layer_model_layers_17/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/mean":8.792267180979252e-08,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/mean":4.824250936508179e-06,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/std":0.0279541015625,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_up_proj/mean":-0.07177734375,"train/train/tensor_act_model_layers_51_post_attention_layernorm/mean":0.04522705078125,"train/train/tensor_act_model_layers_69_self_attn_o_proj/max_abs":1.6171875,"train/train/tensor_act_model_layers_82_post_attention_layernorm/max_abs":5.4375,"train/train/tensor_act_model_layers_53_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/mean":9.105860954150558e-08,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_down_proj/std":0.09643587703889751,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/std":0.0269775390625,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/mean":-6.18956983089447e-06,"train/train/tensor_act_model_layers_0_self_attn_q_proj/max_abs":6.34375,"train/train/tensor_act_model_layers_24_post_attention_layernorm/mean":0.03277587890625,"train/train/tensor_act_model_layers_71_self_attn_q_proj/max_abs":9.1875,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/std":0.044921875,"train/train/tensor_act_model_layers_14/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_input_layernorm/norm":5792.614013675759,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/std":0.00016461456224413509,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/norm":4.09375,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_78/grad/mean":1.9746784603167993e-06,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/std":0.029541015625,"train/train/tensor_act_model_layers_6_mlp_down_proj/std":0.05487070889829195,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp/std":0.053161730585522024,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/max_abs":0.0004119873046875,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/max_abs":0.00107574462890625,"train/train/tensor_act_model_layers_91_input_layernorm/mean":0.06109619140625,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/norm":0.01785818165073722,"train/train/tensor_act_model_layers_57_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/std":0.04833984375,"train/train/tensor_act_model_layers_75_mlp_up_proj/mean":-0.089599609375,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/max_abs":0.1103515625,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/mean":6.280839443206787e-06,"train/train/tensor_param_model_layers_66_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_32/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/std":0.045654296875,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/max_abs":0.00010776519775390625,"train/train/global/act/norm":181671.89971310677,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/norm":6.71875,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/max_abs":0.197265625,"train/train/tensor_act_model_layers_43_self_attn/mean":0.00021916627883911133,"train/train/tensor_act_model_layers_64_self_attn_v_proj/max_abs":3.234375,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_42/grad/max_abs":0.0020599365234375,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/std":0.0250244140625,"train/train/tensor_act_model_layers_60_self_attn_q_proj/mean":0.013519287109375,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/max_abs":0.000507354736328125,"train/train/tensor_act_model_layers_74_mlp_up_proj/mean":-0.0863037109375,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/mean":0.000339508056640625,"train/train/tensor_act_model_layers_31_self_attn/max_abs":0.98046875,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/norm":4.28125,"train/train/tensor_act_model_layers_49/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/mean":0.0001049041748046875,"train/train/tensor_act_model_layers_42/std":1.2949383994956918,"train/train/tensor_act_model_layers_67_self_attn/max_abs":1.3359375,"train/train/tensor_act_model_layers_62_self_attn_k_proj/mean":0.0931396484375,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/norm":0.0031081801250038743,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/max_abs":0.000370025634765625,"train/train/layer_model_layers_83/grad/std":0.00010078194599000284,"train/train/tensor_act_model_layers_14_self_attn_o_proj/norm":297.7368160571121,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp_down_proj/mean":-0.010528564453125,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_63/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/max_abs":0.00026702880859375,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/max_abs":0.166015625,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/grad_norm":1.1171875,"train/train/layer_model_layers_8/act/max_abs":18.5,"train/train/tensor_act_model_layers_7_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp_up_proj/std":0.27734375,"train/train/tensor_act_model_layers_44_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn/mean":-0.0008916854858398438,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/std":5.7565268653640657e-05,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87/std":2.027359038131783,"train/train/tensor_act_model_layers_49_self_attn_o_proj/std":0.08874831462803683,"train/train/layer_model_layers_40/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/max_abs":0.00057220458984375,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_24_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_61/param/mean":0.001442991069252145,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn/std":0.10986562729861304,"train/train/layer_model_layers_1/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_input_layernorm/std":1.0000000553773118,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/std":0.052978515625,"train/train/tensor_act_model_layers_48_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_63/grad/norm":0.05913403668970808,"train/train/layer_model_layers_48/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp/std":0.047485515819745985,"train/train/tensor_act_model_layers_36_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/norm":6.1875,"train/train/tensor_act_model_layers_41_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp_down_proj/mean":-0.003841400146484375,"train/train/layer_model_layers_63/act/norm":14853.193461585483,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_44/param/std":0.05248239175668478,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_47_post_attention_layernorm/std":0.9960938322777808,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/std":0.024658203125,"train/train/layer__model_layers_68/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_norm_weight/std":0,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/std":0.0291748046875,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/norm":6.1875,"train/train/layer__model_layers_25/param/norm":20.185637491481412,"train/train/tensor_act_model_layers_27_self_attn/norm":488.02966710732926,"train/train/tensor_act_model_layers_74_mlp_down_proj/max_abs":1.703125,"train/train/tensor_act_model_layers_21_mlp_down_proj/std":0.040527346622512894,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/max_abs":0.0002002716064453125,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/std":4.2567810427919076e-05,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/norm":0.0004075973235724902,"train/train/layer__model_layers_57/param/max_abs":1,"train/train/tensor_act_model_layers_17_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/norm":5.875,"train/train/tensor_act_model_layers_44_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_input_layernorm/mean":-0.00690460205078125,"train/train/tensor_act_model_layers_68_self_attn_o_proj/mean":0.00189208984375,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/std":0.0390625,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/std":3.0137137868271202e-05,"train/train/layer_model_layers_76/act/max_abs":12.625,"train/train/tensor_param_model_layers_40_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_62_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/mean":-1.457519829273224e-07,"train/train/tensor_act_model_layers_34_self_attn/mean":-0.0007953643798828125,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/norm":3.375,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/std":0.00012153538897899122,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/norm":0.0008557727278550096,"train/train/tensor_act_model_layers_41_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_post_attention_layernorm/norm":5792.607421875947,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/max_abs":0.0005950927734375,"train/train/layer__model_layers_17/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/mean":-3.546476364135742e-06,"train/train/tensor_act_model_layers_76_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/max_abs":0.2265625,"train/train/layer_model_layers_88/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp_up_proj/max_abs":3.765625,"train/train/layer__model_layers_53/param/mean":0.0015216892855401717,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/std":0.031494140625,"train/train/tensor_act_model_layers_17_self_attn/max_abs":0.349609375,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_input_layernorm/max_abs":6.125,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/std":5.6539870624607464e-05,"train/train/layer__model_layers_72/param/std":0.056543745881855345,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/max_abs":0.0001049041748046875,"train/train/tensor_param_model_layers_35_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp_down_proj/max_abs":1.3359375,"train/train/tensor_act_model_layers_75_self_attn_v_proj/max_abs":3.96875,"train/train/tensor_act_model_layers_36_input_layernorm/norm":5792.605102540774,"train/train/tensor_act_model_layers_84_post_attention_layernorm/std":0.9960959640179132,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/max_abs":0.12890625,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/max_abs":0.1591796875,"train/train/tensor_act_model_layers_24_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/std":3.0434256927361175e-05,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/mean":-0.0004405975341796875,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/norm":0.03412099976900605,"train/train/tensor_act_model_layers_16_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp_down_proj/mean":-0.003368377685546875,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/norm":0.0005225090980614641,"train/train/layer_model_layers_18/grad/frac_near_user_limit":0,"train/train/layer_model_layers_45/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/mean":2.504093572497368e-07,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/std":0.037841796875,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/mean":-8.083879947662354e-07,"train/train/tensor_act_model_layers_36_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_6/grad/max_abs":0.00164031982421875,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn_q_proj/mean":0.04412841796875,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/max_abs":0.00015163421630859375,"train/train/tensor_act_model_layers_35_self_attn_v_proj/std":0.3457031760859334,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn/mean":-0.0008215904235839844,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/std":2.5847271509631372e-05,"train/train/layer_model_layers_55/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/mean":5.8710575103759766e-06,"train/train/layer_model_layers_6/act/mean":-0.014111592219426082,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/norm":0.010920390302408491,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/std":7.617824662318272e-05,"train/train/tensor_act_model_layers_76_self_attn_k_proj/std":0.8359376024977007,"train/train/tensor_act_model_layers_36_self_attn/max_abs":1.1015625,"train/train/tensor_act_model_layers_45/norm":7605.641064866444,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/max_abs":0.154296875,"train/train/tensor_act_model_layers_52_self_attn_v_proj/max_abs":2.390625,"train/train/tensor_act_model_layers_85_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_42_self_attn_o_proj/max_abs":1.375,"train/train/tensor_act_model_layers_78_mlp_down_proj/mean":0.018646240234375,"train/train/tensor_act_model_layers_24_mlp_down_proj/mean":0.0012454986572265625,"train/train/layer_model_layers_34/grad/mean":-4.541483386630387e-07,"train/train/tensor_act_model_layers_49_self_attn_v_proj/mean":0.0013275146484375,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/std":0.0277099609375,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/mean":9.052455425262451e-07,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/norm":2.890625,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/mean":4.013418219983578e-08,"train/train/layer_model_layers_9/grad/std":7.100113523653724e-05,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/mean":-3.4247932489961386e-06,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/max_abs":0.2275390625,"train/train/layer_model_layers_65/grad/max_abs":0.00128173828125,"train/train/tensor_act_model_layers_71_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/mean":-2.516433596611023e-06,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/mean":0.000270843505859375,"train/train/tensor_act_model_layers_59_mlp_down_proj/max_abs":1.4921875,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/mean":0.0003032684326171875,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/norm":5.3125,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/mean":0.00020599365234375,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/mean":6.92903995513916e-06,"train/train/tensor_act_model_layers_50_post_attention_layernorm/mean":0.049560546875,"train/train/tensor_act_model_layers_45_self_attn_v_proj/std":0.36523459325811936,"train/train/tensor_act_model_layers_20_mlp/norm":217.8237557133209,"train/train/tensor_act_model_layers_70_self_attn/max_abs":5.53125,"train/train/layer_model_layers_55/act/max_abs":14.875,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/max_abs":0.00140380859375,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/mean":-1.0076910257339478e-06,"train/train/tensor_act_model_layers_71_self_attn_o_proj/max_abs":0.9453125,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn_q_proj/max_abs":5.75,"train/train/tensor_act_model_layers_24_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp/mean":0.002197265625,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/mean":0.000736236572265625,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/mean":0.0002079010009765625,"train/train/tensor_act_model_layers_9_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_5/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/norm":0.012839996733884158,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/std":5.8121223445344274e-05,"train/train/tensor_act_model_layers_13_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/std":0.025146484375,"train/train/tensor_act_model_layers_90_input_layernorm/std":1.0000021066494449,"train/train/tensor_act_model_layers_82_self_attn_v_proj/std":0.42138779139365395,"train/train/tensor_act_model_layers_79_mlp_down_proj/std":0.1977545167183578,"train/train/tensor_act_model_layers_35_input_layernorm/max_abs":6.1875,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/max_abs":0.00051116943359375,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/norm":0.0011144793279383985,"train/train/layer_model_layers_55/grad/norm":0.054282164414073346,"train/train/tensor_act_model_layers_83_self_attn_v_proj/std":0.4843750881451673,"train/train/tensor_param_model_layers_51_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/std":8.639234943522919e-05,"train/train/layer__model_layers_33/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/std":0.038330078125,"train/train/tensor_act_model_layers_69_mlp_down_proj/norm":972.8257132627101,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/norm":0.01883258513494806,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/mean":-4.534376785159111e-08,"train/train/layer_model_layers_83/grad/max_abs":0.001434326171875,"train/train/layer_model_layers_0/act/norm":20040.017016461647,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_q_proj/mean":-0.05548095703125,"train/train/layer_model_layers_29/act/std":0.661892422175934,"train/train/tensor_act_model_layers_55/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/mean":-1.7799437046051025e-05,"train/train/tensor_act_model_layers_10_self_attn/norm":392.23420021884203,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_18_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/std":1.2312727543550562e-05,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/mean":4.810863174498081e-08,"train/train/tensor_act_model_layers_67_post_attention_layernorm/std":1.0000009015198459,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/mean":1.6260601114481688e-06,"train/train/tensor_act_model_layers_56_self_attn_v_proj/max_abs":2.546875,"train/train/tensor_act_model_layers_0_self_attn_k_proj/norm":4289.195607495507,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/norm":0.015733846390376567,"train/train/tensor_param_model_layers_34_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/mean":7.115304470062256e-07,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/norm":0.12841054220445908,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/std":8.282647556605633e-05,"train/train/tensor_act_model_layers_86_self_attn_k_proj/std":1.0839900042413484,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31/std":1.3105629523438094,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/max_abs":0.000701904296875,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/norm":0.016464905110852354,"train/train/tensor_act_model_layers_46_mlp/mean":0.00310516357421875,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51/mean":0.05181884765625,"train/train/tensor_act_model_layers_2_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_34/grad/norm":0.040565014689839494,"train/train/tensor_act_model_layers_80_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/norm":3.578125,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_input_layernorm/max_abs":5.34375,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/std":0.056396484375,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/mean":-1.8451828509569168e-08,"train/train/tensor_act_model_layers_30_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/std":0.03173828125,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/mean":-0.000335693359375,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/std":0.0302734375,"train/train/tensor_act_model_layers_81_self_attn_q_proj/mean":0.064697265625,"train/train/tensor_act_model_layers_64_mlp/max_abs":1.2890625,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/max_abs":0.00020599365234375,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/norm":0.0023662195382648733,"train/train/tensor_param_model_layers_15_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_57_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/std":5.8624064132930086e-05,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/mean":9.441375732421875e-05,"train/train/layer_model_layers_59/act/norm":14001.25489727334,"train/train/tensor_act_model_layers_52_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/norm":0.02424087129413432,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_60_mlp_up_proj/std":0.4316407630885667,"train/train/tensor_act_model_layers_50_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_mlp/norm":483.783498865299,"train/train/tensor_act_model_layers_22_mlp_down_proj/max_abs":0.5390625,"train/train/tensor_act_model_layers_80_self_attn_o_proj/mean":-0.003803253173828125,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/std":0.031494140625,"train/train/tensor_act_model_layers_57_self_attn/max_abs":1.1953125,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/max_abs":0.25390625,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/max_abs":0.271484375,"train/train/layer__model_layers_59/param/mean":0.0015119769829855694,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/mean":5.96512109041214e-07,"train/train/tensor_act_model_layers_29_mlp_down_proj/max_abs":0.98046875,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/mean":1.0989606380462646e-05,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/mean":-1.2256205081939697e-06,"train/train/layer_model_layers_38/grad/mean":-5.338463406318249e-07,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/std":0.032958984375,"train/train/layer_model_layers_54/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/std":9.164701404861528e-06,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/max_abs":0.1435546875,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/std":0.0001203040860854325,"train/train/tensor_act_model_layers_50_self_attn_q_proj/std":0.9062500390513182,"train/train/tensor_act_model_layers_70/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/mean":0.000209808349609375,"train/train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_43/grad/mean":-6.444999212965765e-07,"train/train/tensor_act_model_layers_68_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/mean":0.0009765625,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn/norm":1044.2974707242477,"train/train/tensor_act_model_layers_17_input_layernorm/mean":0.0157318115234375,"train/train/tensor_act_model_layers_2_self_attn_o_proj/max_abs":0.765625,"train/train/tensor_act_model_layers_41_self_attn/mean":-0.0002765655517578125,"train/train/layer_model_layers_23/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/max_abs":0.000850677490234375,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_rotary_emb/mean":0.333984375,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/mean":-3.028661012649536e-06,"train/train/tensor_param_model_layers_73_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_77_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/norm":4.65625,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/norm":4.28125,"train/train/tensor_act_model_layers_17_post_attention_layernorm/max_abs":6.125,"train/train/tensor_act_model_layers_56_self_attn_v_proj/norm":1978.0565081234881,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_q_proj/std":0.8818426691687612,"train/train/tensor_act_model_layers_74_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_59/act/std":0.6704326124529199,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/norm":7.03125,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/max_abs":0.0002574920654296875,"train/train/layer_model_layers_58/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_embed_tokens/mean":0.0016880035400390625,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/mean":0.00014972686767578125,"train/train/layer_model_layers_47/act/max_abs":15.9375,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/max_abs":0.00066375732421875,"train/train/tensor_act_model_layers_17_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/norm":0.005199419778653268,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/norm":7.0625,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/std":0.05615234375,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/std":3.762377950518601e-05,"train/train/tensor_act_model_layers_39_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/norm":5.65625,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_75/act/std":0.7580015911293548,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/std":5.716209204321875e-05,"train/train/tensor_act_model_layers_39_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/std":0.0361328125,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/norm":0.004961429931951243,"train/train/tensor_act_model_layers_24_input_layernorm/mean":0.03302001953125,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/norm":0.016388024697534927,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/mean":-1.0260919225402176e-07,"train/train/layer_model_layers_3/grad/mean":-3.0507189011006573e-07,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/norm":0.01651429633657984,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/max_abs":0.171875,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/max_abs":0.2138671875,"train/train/tensor_act_model_layers_19_self_attn_v_proj/norm":1351.3183199144491,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_o_proj/norm":1401.4575336815262,"train/train/tensor_act_model_layers_65_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_63_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_74/act/std":0.7303653767973389,"train/train/tensor_act_model_layers_44_self_attn_q_proj/mean":-0.0516357421875,"train/train/tensor_act_model_layers_35_self_attn_k_proj/max_abs":4.9375,"train/train/tensor_act_model_layers_10_self_attn_q_proj/mean":-0.04144287109375,"train/train/tensor_act_model_layers_6_self_attn_v_proj/std":0.29199420119312064,"train/train/tensor_act_model_layers_82_self_attn_v_proj/norm":2441.320915795167,"train/train/tensor_act_model_layers_19_self_attn_k_proj/std":0.7959007590792444,"train/train/tensor_act_model_layers_84/norm":10690.14796635889,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/norm":6.96875,"train/train/layer_model_layers_53/act/std":0.6623873442129983,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/mean":-2.419576048851013e-06,"train/train/layer_model_layers_58/act/max_abs":14.4375,"train/train/tensor_act_model_layers_31_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp_down_proj/std":0.060302740166544154,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/mean":-1.89291313290596e-07,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_o_proj/std":0.17359298616151067,"train/train/tensor_act_model_layers_33_post_attention_layernorm/max_abs":6.3125,"train/train/tensor_act_model_layers_19_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn/norm":303.15854645793576,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/norm":5.5,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/std":0.00010326550786277409,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_post_attention_layernorm/norm":5792.602416993004,"train/train/tensor_param_model_layers_53_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52/mean":0.0546875,"train/train/tensor_act_model_layers_81_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn/norm":690.4142818059483,"train/train/layer_model_layers_10/act/mean":-0.01874512892503005,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_up_proj/std":0.2529316905763468,"train/train/tensor_act_model_layers_57/std":1.324224702714802,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/norm":0.002917053877794694,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/norm":5.28125,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/mean":8.821487426757812e-05,"train/train/layer_model_layers_90/act/std":0.9441709584316438,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/max_abs":0.000232696533203125,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn_k_proj/norm":4529.166133399788,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/norm":8.75,"train/train/tensor_param_model_layers_13_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn_v_proj/std":0.30468752185025966,"train/train/layer_model_layers_40/grad/norm":0.044748875727484747,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/std":0.033447265625,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/max_abs":0.1689453125,"train/train/tensor_act_model_layers_91_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_input_layernorm/mean":0.00420379638671875,"train/train/tensor_act_model_layers_43_self_attn_k_proj/max_abs":4.71875,"train/train/layer__model_layers_89/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_20/act/max_abs":17.875,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/norm":0.015460820718548778,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp/mean":0.0016613006591796875,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/norm":0.005180959989704965,"train/train/tensor_act_model_layers_47/norm":7588.186455359821,"train/train/tensor_act_model_layers_25_self_attn/mean":-0.003017425537109375,"train/train/tensor_act_model_layers_15/mean":0.01654052734375,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_82/act/norm":16415.101396661677,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/mean":-1.2675300240516663e-06,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/mean":-4.604458808898926e-06,"train/train/tensor_act_model_layers_87_self_attn_k_proj/norm":5657.92389709634,"train/train/tensor_act_model_layers_51_self_attn/mean":-0.001678466796875,"train/train/layer__model_layers_57/param/std":0.054066536153962695,"train/train/tensor_act_model_layers_10/norm":8356.845438415447,"train/train/tensor_act_model_layers_50_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/max_abs":0.1875,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/mean":3.022141754627228e-07,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/mean":5.052424967288971e-07,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/max_abs":0.09912109375,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/mean":-0.000896453857421875,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/max_abs":0.00119781494140625,"train/train/layer_model_layers_2/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_input_layernorm/norm":5792.607299808601,"train/train/tensor_act_model_layers_78_self_attn_v_proj/std":0.4545911062655347,"train/train/tensor_act_model_layers_32_mlp_up_proj/mean":-0.0718994140625,"train/train/tensor_act_model_layers_52_mlp_down_proj/norm":530.4828839925549,"train/train/tensor_act_model_layers_76_input_layernorm/mean":0.04962158203125,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/std":6.0465409936403383e-05,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_67_self_attn_q_proj/std":0.9482443536444928,"train/train/tensor_act_model_layers_16_mlp_up_proj/mean":-0.0537109375,"train/train/layer_model_layers_14/grad/norm":0.03232552928056038,"train/train/tensor_act_model_layers_13_self_attn_v_proj/max_abs":1.4375,"train/train/tensor_act_model_layers_33_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_37/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_k_proj/mean":-0.045166015625,"train/train/tensor_act_model_layers_11_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp/norm":311.1432610160008,"train/train/tensor_act_model_layers_92_self_attn_q_proj/mean":-0.0020885467529296875,"train/train/tensor_act_model_layers_51/norm":7506.023605931181,"train/train/tensor_act_model_layers_64_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/max_abs":0.232421875,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/std":0.0005016043542413093,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/max_abs":0.1298828125,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_83/param/std":0.05988777424091841,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/max_abs":0.00023651123046875,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/mean":0.0001583099365234375,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/mean":-0.000152587890625,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/max_abs":0.000591278076171875,"train/train/layer_model_layers_7/grad/mean":-2.531824392294177e-07,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_input_layernorm/mean":0.023223876953125,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_14_self_attn_k_proj/max_abs":4.4375,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_53_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13/std":1.4101833857647437,"train/train/layer_model_layers_22/grad/max_abs":0.0016632080078125,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/norm":0.002604765127875369,"train/train/tensor_act_model_layers_28_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_down_proj/std":0.0451660233574938,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/norm":3.21875,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/norm":7.59375,"train/train/tensor_act_model_layers_7_self_attn_v_proj/std":0.31640644573864946,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/mean":5.436595529317856e-08,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/norm":6.5625,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/max_abs":8.249282836914062e-05,"train/train/tensor_act_model_layers_89/mean":0.1031494140625,"train/train/tensor_act_model_layers_25_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_38_self_attn_q_proj/std":0.9648438330121334,"train/train/layer_model_layers_57/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_q_proj/max_abs":6.375,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/mean":-0.00055694580078125,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/mean":1.915759639814496e-08,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/std":9.715966789515075e-05,"train/train/tensor_act_model_layers_25/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_60/act/mean":-0.009868144989013672,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/std":0.044921875,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/norm":0.04682808753779952,"train/train/tensor_act_model_layers_28_self_attn_o_proj/std":0.09069860037621252,"train/train/tensor_act_model_layers_69_mlp/mean":-0.012786865234375,"train/train/tensor_act_model_layers_44_post_attention_layernorm/max_abs":5.84375,"train/train/tensor_act_model_layers_50_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/mean":-1.843273639678955e-05,"train/train/layer_model_layers_10/grad/mean":-3.608083134508356e-07,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/mean":-0.00017452239990234375,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/norm":0.022325271888649304,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/max_abs":0.000148773193359375,"train/train/tensor_param_model_layers_44_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_input_layernorm/max_abs":4.65625,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/max_abs":0.201171875,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/std":4.829876419469162e-05,"train/train/tensor_param_model_layers_6_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_69/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn/max_abs":1.1875,"train/train/layer__model_layers_73/param/std":0.05558100680286242,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/norm":0.00476282025490705,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/norm":4.9375,"train/train/tensor_act_model_layers_72_mlp/max_abs":1.3203125,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/max_abs":0.000240325927734375,"train/train/tensor_act_model_layers_12_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/std":0.0289306640625,"train/train/tensor_act_model_layers_1_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/max_abs":0.11962890625,"train/train/tensor_act_model_layers_52_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/max_abs":0.1953125,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/max_abs":0.000530242919921875,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_q_proj/norm":5890.640060210269,"train/train/tensor_act_model_layers_20/max_abs":17.875,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/mean":-2.130866050720215e-06,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/mean":-0.00015163421630859375,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/max_abs":0.1279296875,"train/train/tensor_act_model_layers_64_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/mean":4.0105078369379044e-08,"train/train/tensor_act_model_layers_10_mlp_down_proj/norm":255.64503033449438,"train/train/tensor_act_model_layers_69_input_layernorm/max_abs":5.84375,"train/train/layer__model_layers_74/param/max_abs":1,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/norm":3.328125,"train/train/tensor_act_model_layers_43_self_attn_o_proj/mean":0.00021916627883911133,"train/train/layer_model_layers_29/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_input_layernorm/max_abs":5.90625,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/mean":6.102025508880615e-06,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/mean":-0.00010538101196289062,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/norm":5.71875,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/mean":0.0003185272216796875,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/norm":0.005146081753061364,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/max_abs":0.00162506103515625,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/std":6.137255209691675e-05,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/std":0.04931640625,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/norm":0.010584057503739936,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/mean":1.1682510375976562e-05,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/std":5.116473375838881e-05,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74/mean":0.0880126953125,"train/train/layer__model_layers_18/param/std":0.04861678179929997,"train/train/tensor_act_model_layers_51_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18/mean":0.02691650390625,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/max_abs":0.00104522705078125,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/max_abs":0.000667572021484375,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_14_mlp/norm":267.99551695116503,"train/train/tensor_act_model_layers_62_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/std":1.0243580116709251e-05,"train/train/tensor_param_model_layers_79_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_post_attention_layernorm/mean":0.0772705078125,"train/train/layer__model_layers_24/param/norm":20.0385919169892,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/max_abs":0.1201171875,"train/train/layer__model_layers_38/param/std":0.051369309416242494,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/mean":0.00012302398681640625,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/mean":-0.00011920928955078125,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/std":0.035400390625,"train/train/tensor_act_model_layers_63_self_attn_q_proj/mean":-0.032470703125,"train/train/tensor_act_model_layers_53_self_attn/mean":0.0011157989501953125,"train/train/layer_model_layers_81/grad/mean":1.5453358475763972e-06,"train/train/tensor_act_model_layers_4_mlp/max_abs":0.53125,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/std":0.029541015625,"train/train/tensor_act_model_layers_62_mlp/mean":0.0016345977783203125,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/mean":2.3085158318281174e-07,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/max_abs":0.002105712890625,"train/train/layer_model_layers_21/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/std":5.873282996807483e-05,"train/train/layer__model_layers_46/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_down_proj/mean":-0.00670623779296875,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/std":4.583033024762411e-05,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_v_proj/norm":2413.8400844495004,"train/train/tensor_act_model_layers_18_mlp/std":0.03643816320099371,"train/train/tensor_act_model_layers_64_mlp/mean":-0.0079803466796875,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/norm":0.06024698622672658,"train/train/tensor_param_model_layers_16_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25/norm":7730.620658669284,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_11_input_layernorm/max_abs":5.75,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_down_proj/max_abs":0.56640625,"train/train/tensor_param_model_layers_29_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_16_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_37/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_k_proj/max_abs":4.90625,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/std":4.924526803850822e-05,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/std":3.643949781111552e-05,"train/train/tensor_act_model_layers_11_mlp_up_proj/max_abs":2.359375,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/act/norm":17831.734710853812,"train/train/tensor_act_model_layers_20_self_attn_q_proj/max_abs":6.3125,"train/train/tensor_act_model_layers_50_self_attn_k_proj/mean":0.028167724609375,"train/train/layer_model_layers_59/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_v_proj/std":0.2429203948778311,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/max_abs":0.0001544952392578125,"train/train/tensor_act_model_layers_89_self_attn_k_proj/std":1.0371148935223253,"train/train/tensor_param_model_layers_76_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp/mean":0.009033203125,"train/train/tensor_act_model_layers_65_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn/norm":1387.4049382221137,"train/train/tensor_act_model_layers_19_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/max_abs":0.140625,"train/train/layer_model_layers_14/grad/max_abs":0.0012969970703125,"train/train/tensor_act_model_layers_19_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp_down_proj/mean":-0.00921630859375,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/norm":6.96875,"train/train/tensor_act_model_layers_21_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_63/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_84_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp/max_abs":0.55859375,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/norm":6.5,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/max_abs":0.1279296875,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/max_abs":0.000370025634765625,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_15/grad/mean":-3.927150107745819e-07,"train/train/layer_model_layers_32/act/std":0.6578251556232674,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/norm":3.0625,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/max_abs":7.581710815429688e-05,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_42_self_attn_o_proj/mean":-0.002109527587890625,"train/train/tensor_act_model_layers_74_self_attn_k_proj/max_abs":4.96875,"train/train/layer__model_layers_59/param/std":0.053739816184727555,"train/train/tensor_act_model_layers_5_self_attn_q_proj/mean":0.0831298828125,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/norm":3.546875,"train/train/tensor_act_model_layers_46_self_attn_q_proj/std":1.0468750556013462,"train/train/tensor_act_model_layers_27_input_layernorm/std":1.0000000223517413,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/norm":0.05548132632964619,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/norm":0.012124430618484991,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_74/act/norm":15245.96935332555,"train/train/tensor_act_model_layers_23/mean":0.0379638671875,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_o_proj/std":0.13940500021524083,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/std":0.032470703125,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/std":2.560737246268493e-05,"train/train/layer_model_layers_76/grad/norm":0.07103159314902437,"train/train/tensor_act_model_layers_51_post_attention_layernorm/max_abs":5.8125,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/std":4.1344827433658295e-05,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/std":0.026611328125,"train/train/tensor_act_model_layers_20_self_attn_v_proj/mean":0.0009708404541015625,"train/train/tensor_act_model_layers_7_self_attn_k_proj/norm":9122.765511368321,"train/train/tensor_act_model_layers_31_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/std":7.218891391515922e-05,"train/train/epoch_time_elapsed":3736.818679008633,"train/train/tensor_act_model_layers_68_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_49/grad/frac_near_dtype_limit":0,"train/train/layer__model_layers_91/param/mean":0.0013432971400887286,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/max_abs":0.154296875,"train/train/tensor_act_model_layers_69_post_attention_layernorm/mean":0.03814697265625,"train/train/layer_model_layers_32/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_k_proj/max_abs":5.71875,"train/train/layer_model_layers_13/grad/mean":-4.172027285404027e-07,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/max_abs":0.2392578125,"train/train/tensor_act_model_layers_58/norm":7682.6234450939155,"train/train/tensor_act_model_layers_46_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn_v_proj/std":0.42187504855380975,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/norm":0.004997876543405018,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/max_abs":0.205078125,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/std":8.43697239517482e-05,"train/train/tensor_act_model_layers_17_self_attn_v_proj/norm":1395.792441863175,"train/train/tensor_act_model_layers_66_self_attn_k_proj/mean":-0.0250244140625,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_14/act/norm":14425.948342505217,"train/train/tensor_act_model_layers_10_post_attention_layernorm/std":1.0000000083819032,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/mean":1.187250018119812e-05,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/norm":4.46875,"train/train/layer__model_layers_41/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/std":7.944058011780621e-05,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/mean":-0.0001277923583984375,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/norm":6,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_61/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_k_proj/norm":4823.556514374229,"train/train/layer__model_layers_52/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp_down_proj/std":0.22509820200032668,"train/train/tensor_act_model_layers_83_self_attn_o_proj/max_abs":1.8515625,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_77/param/max_abs":1,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/mean":9.424984455108643e-07,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/std":4.995234086643003e-05,"train/train/tensor_act_model_layers_5_mlp_down_proj/max_abs":0.515625,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/max_abs":0.1220703125,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_v_proj/norm":1982.4600291235652,"train/train/tensor_act_model_layers_1_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn_q_proj/std":0.9121115037576208,"train/train/layer__model_layers_24/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/std":2.3430403894130022e-05,"train/train/tensor_act_model_layers_22_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3/mean":0.0157318115234375,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_76/grad/std":8.763654983560215e-05,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/mean":-0.00018978118896484375,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/max_abs":0.00049591064453125,"train/train/tensor_act_model_layers_53_post_attention_layernorm/max_abs":5.875,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/std":5.454526885430699e-05,"train/train/tensor_act_model_layers_54_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp/mean":-0.003173828125,"train/train/tensor_act_model_layers_14_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_23/grad/max_abs":0.0019378662109375,"train/train/tensor_act_model_layers_12_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/std":0.0341796875,"train/train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/max_abs":0.09619140625,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/norm":5.9375,"train/train/tensor_act_model_rotary_emb/max_abs":1,"train/train/tensor_act_model_layers_30/std":1.3164352900530845,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/max_abs":0.0005645751953125,"train/train/layer_model_layers_10/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp_up_proj/max_abs":6.5625,"train/train/tensor_act_model_layers_50/mean":0.05670166015625,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_43/act/mean":-0.006108238146855281,"train/train/tensor_act_model_layers_56_self_attn/std":0.05646045046525053,"train/train/tensor_act_model_layers_34_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_0/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/norm":5.34375,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/max_abs":0.00010585784912109375,"train/train/tensor_act_model_layers_51_mlp_up_proj/std":0.37939584699915624,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/mean":-0.00018215179443359375,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/mean":-1.0147690773010254e-05,"train/train/tensor_act_model_layers_30_self_attn_k_proj/norm":5798.759425757493,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/mean":-0.000789642333984375,"train/train/tensor_act_model_layers_39_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/max_abs":0.0004730224609375,"train/train/tensor_act_model_layers_58_mlp/mean":-0.0025482177734375,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/norm":0.03648432304819725,"train/train/tensor_act_model_layers_21_mlp_up_proj/std":0.23168993548874642,"train/train/tensor_act_model_layers_8_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_26/act/max_abs":17.75,"train/train/layer_model_layers_11/act/norm":14712.014244474702,"train/train/tensor_act_model_layers_22_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/norm":0.011728600494560494,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_26_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_65/act/max_abs":13.4375,"train/train/tensor_act_model_layers_74_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/mean":8.630752563476562e-05,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/norm":0.029828649565219065,"train/train/tensor_act_model_layers_12_mlp_down_proj/mean":0.003559112548828125,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/std":0.051513671875,"train/train/tensor_param_model_layers_27_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/std":0.02685546875,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/mean":0.00014400482177734375,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/mean":-0.00019168853759765625,"train/train/tensor_act_model_layers_70/std":1.4824280405733057,"train/train/layer__model_layers_29/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_input_layernorm/mean":0.0438232421875,"train/train/tensor_act_model_layers_29_input_layernorm/std":1.0000000447034827,"train/train/tensor_act_model_layers_28_self_attn_k_proj/norm":5572.5886840745725,"train/train/tensor_act_model_layers_93_self_attn_k_proj/mean":0.03558349609375,"train/train/tensor_act_model_layers_12_mlp_up_proj/norm":2157.068964250751,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/max_abs":0.000560760498046875,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_91/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/std":0.037109375,"train/train/tensor_param_model_layers_82_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/mean":5.125999450683594e-05,"train/train/tensor_act_model_layers_8_self_attn/frac_near_dtype_limit":0,"train/train/layer__model_layers_30/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/std":6.308120805873555e-05,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/max_abs":0.0009918212890625,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_29/grad/max_abs":0.00150299072265625,"train/train/layer_model_layers_52/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/max_abs":0.000125885009765625,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/max_abs":0.0010986328125,"train/train/tensor_act_model_layers_36_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/mean":0.0002422332763671875,"train/train/tensor_act_model_layers_4_mlp/frac_near_dtype_limit":0,"train/train/layer__model_layers_50/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_9/grad/norm":0.057523526880680034,"train/train/tensor_act_model_layers_59_mlp_up_proj/norm":4279.9398600567165,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn_k_proj/mean":-0.027740478515625,"train/train/tensor_act_model_layers_0_post_attention_layernorm/max_abs":5.28125,"train/train/layer__model_layers_42/param/mean":0.0014527628834645574,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/mean":1.935986801981926e-07,"train/train/tensor_act_model_layers_12_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_q_proj/norm":6807.996750997039,"train/train/layer_model_layers_13/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/norm":6.3125,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_v_proj/norm":2375.659358503556,"train/train/layer_model_layers_39/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/std":1.0000000166387506,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/norm":0.0010808677293072373,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/mean":-2.9966235160827637e-05,"train/train/layer__model_layers_76/param/norm":23.014325531883397,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/max_abs":0.00012969970703125,"train/train/layer_model_layers_36/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/std":9.735139908197954e-05,"train/train/tensor_act_model_layers_20_self_attn_o_proj/max_abs":0.48046875,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/std":0.023681640625,"train/train/layer_model_layers_64/grad/mean":-7.843029178248925e-07,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_input_layernorm/norm":5792.613037109956,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/mean":0.00025177001953125,"train/train/tensor_act_model_layers_54/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp_up_proj/mean":-0.055908203125,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_input_layernorm/mean":0.0589599609375,"train/train/tensor_act_model_layers_74_post_attention_layernorm/max_abs":5.4375,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/std":0.029541015625,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/norm":0.0005370499025277243,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/mean":3.442983143031597e-07,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/std":1.4879712250407378e-05,"train/train/tensor_act_model_layers_5_self_attn_k_proj/mean":0.0699462890625,"train/train/tensor_act_model_layers_71_self_attn_k_proj/mean":0.03338623046875,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_50/grad/norm":0.03845267134643921,"train/train/layer_model_layers_3/act/frac_near_dtype_limit":0,"train/train/layer__model_layers_66/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/std":0.0303955078125,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/mean":1.6880221664905548e-09,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/mean":5.602836608886719e-05,"train/train/tensor_act_model_layers_0_mlp/norm":8421.160495516604,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/norm":7.1875,"train/train/tensor_act_model_layers_92_self_attn_v_proj/norm":2838.5375503761397,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/mean":-0.00063323974609375,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/mean":-2.012820914387703e-07,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/norm":0.0008840926120625907,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/max_abs":0.00012874603271484375,"train/train/tensor_act_model_layers_59_self_attn_k_proj/mean":-0.0384521484375,"train/train/tensor_act_model_layers_79_self_attn/max_abs":2.40625,"train/train/layer_model_layers_36/act/max_abs":17.375,"train/train/layer__model_layers_34/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/mean":0.00012683868408203125,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_68_self_attn_q_proj/max_abs":5.3125,"train/train/tensor_act_model_layers_19_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/norm":0.0025207962881997046,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_47/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/norm":6.3125,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/std":5.4155667116025737e-05,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_28/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23/std":1.3554969202022118,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/mean":1.2543750926852226e-07,"train/train/tensor_act_model_layers_68/std":1.4589900383079744,"train/train/tensor_act_model_layers_61_mlp_up_proj/max_abs":3.71875,"train/train/tensor_act_model_layers_54_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/std":0.035888671875,"train/train/tensor_act_model_layers_57_self_attn_o_proj/max_abs":1.1953125,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/norm":4.75,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/mean":2.5425106287002563e-07,"train/train/tensor_act_model_layers_24_self_attn_k_proj/norm":5500.681358114986,"train/train/tensor_act_model_layers_92_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_69/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/norm":3.1875,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/max_abs":0.275390625,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/norm":3.484375,"train/train/layer__model_layers_24/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41/max_abs":16.875,"train/train/tensor_act_model_layers_26_mlp_down_proj/std":0.047485515819745985,"train/train/layer_model_layers_76/grad/max_abs":0.00159454345703125,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/mean":-3.620516508817673e-08,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_16/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/std":2.3850723585874355e-05,"train/train/tensor_act_model_layers_39_self_attn_o_proj/std":0.06207425511598431,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/std":0.045166015625,"train/train/layer_model_layers_83/grad/mean":1.3493512889338359e-06,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/mean":-4.291534423828125e-05,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/std":0.00011296110244625884,"train/train/layer_model_layers_10/grad/norm":0.03830669182805127,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/norm":0.003983281969695029,"train/train/tensor_act_model_layers_22/max_abs":17.75,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/std":3.0085089172351815e-05,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/norm":146.8135230071949,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/max_abs":0.1611328125,"train/train/tensor_act_model_layers_74_input_layernorm/max_abs":5.5,"train/train/tensor_act_model_layers_8_mlp_down_proj/mean":-0.00046443939208984375,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/std":0.0283203125,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/mean":3.214925527572632e-06,"train/train/tensor_act_model_layers_93_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer__model_layers_42/param/norm":20.816916057394884,"train/train/layer_model_layers_50/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_q_proj/max_abs":6.03125,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/norm":0.0016935666327593604,"train/train/tensor_param_model_layers_71_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_61/std":1.3515739826309978,"train/train/tensor_act_model_layers_87/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_v_proj/mean":0.0012493133544921875,"train/train/tensor_act_model_layers_93_mlp_up_proj/norm":9527.797468221299,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_self_attn/norm":326.51071662888035,"train/train/tensor_act_model_layers_23_self_attn/std":0.11926654774189237,"train/train/tensor_act_model_layers_32_self_attn_o_proj/norm":238.1257958749373,"train/train/layer_model_layers_46/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_83/act/norm":17306.307919843417,"train/train/tensor_act_model_layers_29_self_attn_v_proj/mean":-0.00479888916015625,"train/train/layer__model_layers_32/param/max_abs":1,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_k_proj/std":1.111333367682294,"train/train/tensor_act_model_layers_44_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_mlp/norm":298.93819581491954,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/max_abs":0.1689453125,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/max_abs":7.343292236328125e-05,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/max_abs":0.23046875,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/std":8.824646473557116e-06,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/mean":0.000438690185546875,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/mean":-2.8207432478666306e-07,"train/train/tensor_act_model_layers_64_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp/norm":353.5603615532615,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/norm":6.84375,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/mean":0.000560760498046875,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/norm":4.46875,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/std":9.096135828201887e-05,"train/train/tensor_act_model_layers_92_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_79/act/mean":-0.018518594595102165,"train/train/tensor_act_model_layers_40_self_attn_o_proj/norm":536.6007892014339,"train/train/tensor_param_model_layers_39_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/max_abs":0.00148773193359375,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/norm":7.5,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/max_abs":0.2080078125,"train/train/tensor_act_model_layers_79/std":1.6855545264785392,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/mean":-0.0002079010009765625,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/norm":0.002307333760526976,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/std":0.0001114869691803123,"train/train/tensor_param_model_layers_60_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_89/act/max_abs":12.3125,"train/train/tensor_act_model_layers_83_mlp_up_proj/max_abs":3.890625,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/norm":0.013003138598870664,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/std":9.42755340346365e-05,"train/train/tensor_act_model_layers_6_mlp/std":0.05487070889829195,"train/train/tensor_act_model_layers_0_mlp_up_proj/std":0.9599651965243761,"train/train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_mlp_up_proj/norm":4768.657368817339,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/max_abs":0.232421875,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/std":0.056640625,"train/train/tensor_act_model_layers_49_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/mean":-0.00016117095947265625,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/norm":6.3125,"train/train/tensor_act_model_layers_38_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/norm":0.002259099664373176,"train/train/tensor_act_model_layers_11_self_attn_k_proj/mean":-0.063720703125,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/norm":3.203125,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_input_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_25/param/max_abs":1,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/norm":0.0025394081834079767,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_4/act/max_abs":18.5,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/mean":0.000640869140625,"train/train/tensor_act_model_layers_30_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/norm":0.00038453302668853127,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/std":3.566764127011393e-05,"train/train/tensor_act_model_layers_18_input_layernorm/std":1.0000000670552232,"train/train/tensor_act_model_layers_47_post_attention_layernorm/norm":5792.607421875689,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/std":0.0233154296875,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/mean":1.153675839304924e-07,"train/train/layer__model_layers_11/param/max_abs":1,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/mean":-8.273124694824219e-05,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_v_proj/mean":-0.0090789794921875,"train/train/tensor_act_model_layers_68_self_attn/max_abs":2.46875,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/max_abs":0.00174713134765625,"train/train/layer_model_layers_55/grad/mean":-1.3765136532590095e-06,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/max_abs":0.00057220458984375,"train/train/tensor_param_model_layers_53_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_k_proj/std":0.8134787368412627,"train/train/tensor_act_model_layers_86/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_91/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/max_abs":0.259765625,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn_o_proj/std":0.11926654774189237,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/mean":-0.0011444091796875,"train/train/tensor_act_model_layers_15_self_attn_k_proj/norm":6162.863266366149,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn/max_abs":1.2734375,"train/train/layer_model_layers_15/act/norm":14710.480492080647,"train/train/tensor_act_model_layers_75_self_attn_q_proj/norm":5871.240282439001,"train/train/tensor_act_model_layers_71_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/max_abs":0.0001277923583984375,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/norm":0.006616845047746843,"train/train/tensor_act_model_layers_29_self_attn_v_proj/std":0.3046876508264596,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/norm":0.004442221107986416,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/grad/std":4.0987227513909844e-05,"train/train/tensor_act_model_layers_12/std":1.4179957336731484,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/max_abs":0.2470703125,"train/train/layer_model_layers_40/grad/std":5.525475918681201e-05,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/std":3.7435194600506096e-05,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/std":9.823356368998617e-05,"train/train/tensor_act_model_layers_48_post_attention_layernorm/mean":0.0634765625,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_73/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/mean":-0.0003376007080078125,"train/train/tensor_act_model_layers_63/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/mean":-8.812639862298965e-08,"train/train/tensor_act_model_layers_28_mlp_up_proj/max_abs":4.59375,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/norm":0.05238139378890127,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/norm":0.021088242694911493,"train/train/tensor_act_model_layers_13_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_q_proj/mean":0.013946533203125,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60/max_abs":14,"train/train/tensor_act_model_layers_25_post_attention_layernorm/norm":5792.612060555112,"train/train/tensor_act_model_layers_63_self_attn_v_proj/norm":2552.884112286341,"train/train/tensor_act_model_layers_34_self_attn/max_abs":1.2109375,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/mean":2.0791776478290558e-07,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_mlp/max_abs":0.318359375,"train/train/layer_model_layers_6/act/norm":15737.41196543153,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/max_abs":0.00014781951904296875,"train/train/tensor_act_model_layers_91_mlp_up_proj/std":0.8173850001370813,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/max_abs":0.00136566162109375,"train/train/tensor_act_model_layers_93_self_attn_q_proj/std":0.9375000635782856,"train/train/tensor_act_model_layers_29_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/std":3.421999419486647e-05,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/mean":-0.0004119873046875,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/norm":3.15625,"train/train/layer_model_layers_85/grad/norm":0.07932517162027791,"train/train/layer_model_layers_39/act/max_abs":17.25,"train/train/layer_model_layers_35/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/mean":4.243338480591774e-08,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/max_abs":0.0005950927734375,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_69/act/max_abs":13.25,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/mean":5.498528480529785e-06,"eval/samples_per_second":584.844,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/max_abs":0.000270843505859375,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/norm":7,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/max_abs":0.1513671875,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/max_abs":0.0014190673828125,"train/train/layer_model_layers_62/grad/norm":0.058808778796495556,"train/train/tensor_act_model_layers_69_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_o_proj/max_abs":2.46875,"train/train/tensor_act_model_layers_8_mlp/mean":-0.00046443939208984375,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/max_abs":0.142578125,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_33/act/max_abs":17.625,"train/train/tensor_param_model_layers_17_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_34/std":1.3046992638337425,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/max_abs":9.632110595703125e-05,"train/train/tensor_act_model_layers_35_self_attn_o_proj/mean":0.00139617919921875,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/max_abs":0.0001068115234375,"train/train/tensor_act_model_layers_59_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/max_abs":0.00011920928955078125,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/norm":3.453125,"train/train/tensor_act_model_layers_76_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_input_layernorm/mean":0.05316162109375,"train/train/tensor_act_model_layers_49_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_k_proj/max_abs":5.0625,"train/train/tensor_param_model_layers_69_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_59_self_attn/mean":-0.00145721435546875,"train/train/tensor_act_model_layers_37_input_layernorm/max_abs":6.09375,"train/train/tensor_act_model_layers_78_post_attention_layernorm/max_abs":5.75,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/std":0.000799577347865184,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/norm":4.28125,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn/norm":690.7827898320194,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp_down_proj/std":0.12548926367913205,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/norm":5.90625,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/std":6.544728940803814e-05,"train/train/tensor_act_model_layers_65_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/std":0.038818359375,"train/train/tensor_act_model_layers_57_self_attn_q_proj/max_abs":7.125,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/max_abs":0.12353515625,"train/train/tensor_act_model_layers_67_mlp_up_proj/mean":-0.10009765625,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/std":0.033447265625,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/max_abs":0.263671875,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/max_abs":0.0001735687255859375,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/max_abs":0.0009307861328125,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/norm":9.375,"train/train/tensor_act_model_layers_71_mlp_down_proj/max_abs":1.4453125,"train/train/tensor_act_model_layers_16_input_layernorm/norm":5792.609863286014,"train/train/tensor_param_model_layers_27_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp_down_proj/norm":4739.755600762709,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/max_abs":0.000518798828125,"train/train/layer_model_layers_29/act/max_abs":17.875,"train/train/layer_model_layers_71/act/norm":14691.708274958704,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/mean":-3.9493897929787636e-08,"train/train/tensor_act_model_layers_33_input_layernorm/norm":5792.603637696356,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/max_abs":0.11474609375,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/norm":0.021639782031886504,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/std":0.04248046875,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/mean":-8.20159912109375e-05,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/mean":1.5227124094963074e-07,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn_o_proj/std":0.0456547808414322,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/mean":1.1816155165433884e-07,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/mean":-2.6868656277656555e-07,"train/train/tensor_act_model_layers_27_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_mlp_up_proj/std":0.21655331926578636,"train/train/tensor_act_model_layers_48_self_attn_q_proj/mean":0.058837890625,"train/train/layer_model_layers_11/grad/norm":0.032463942489811055,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/norm":0.020538398096388577,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/norm":4.53125,"train/train/tensor_act_model_layers_26_mlp_down_proj/norm":282.2265651657156,"train/train/tensor_act_model_norm/mean":0.058349609375,"train/train/layer_model_layers_24/grad/max_abs":0.00150299072265625,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/mean":-5.329493433237076e-07,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/norm":0.01602553549450582,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/max_abs":0.001434326171875,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/max_abs":0.2119140625,"train/train/tensor_act_model_layers_78_self_attn_k_proj/norm":4910.508648120452,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/mean":-3.300607204437256e-06,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_o_proj/max_abs":1.5234375,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/std":0.032958984375,"train/train/tensor_act_model_layers_90_self_attn_k_proj/norm":4634.562708644993,"train/train/tensor_act_model_layers_42_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/std":0.031982421875,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/std":0.042236328125,"train/train/tensor_act_model_layers_66_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_input_layernorm/std":1.0000023022267543,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_24/param/max_abs":1,"train/train/tensor_act_model_layers_28_input_layernorm/mean":0.04840087890625,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/max_abs":0.00011539459228515625,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/max_abs":0.000148773193359375,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/mean":1.3408134691417217e-07,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/std":0.0224609375,"train/train/tensor_act_model_layers_8_mlp/max_abs":0.341796875,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/max_abs":0.23828125,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_53_self_attn_q_proj/std":0.8964866438693588,"train/train/tensor_act_model_layers_16_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/max_abs":0.0004634857177734375,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/std":0.046630859375,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_88/param/norm":24.794073762695795,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/std":1.2691855403669975e-05,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/std":0.028564453125,"train/train/tensor_act_model_layers_17_self_attn_q_proj/mean":-0.03729248046875,"train/train/tensor_act_model_layers_66_self_attn_k_proj/std":0.8847741932906696,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/norm":5.09375,"train/train/tensor_act_model_layers_14_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_v_proj/max_abs":1.484375,"train/train/layer_model_layers_86/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/mean":0.000446319580078125,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/max_abs":0.10791015625,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/std":0.02685546875,"train/train/tensor_act_model_norm/norm":5792.611938479005,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_up_proj/std":0.3266616844061672,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/max_abs":0.134765625,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/mean":7.666647434234619e-06,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/mean":-5.296897143125534e-08,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/norm":5.09375,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/std":4.226447505735248e-05,"train/train/tensor_act_model_layers_62_self_attn_v_proj/max_abs":3.265625,"train/train/tensor_act_model_layers_26_mlp_up_proj/max_abs":2.59375,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/max_abs":0.181640625,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/norm":5.9375,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/max_abs":0.248046875,"train/train/tensor_act_model_layers_57_input_layernorm/mean":0.04437255859375,"train/train/tensor_act_model_layers_27_self_attn_o_proj/std":0.0842296578061084,"train/train/tensor_act_model_layers_46_mlp_up_proj/norm":3708.9172591324705,"train/train/tensor_act_model_layers_18_self_attn_v_proj/mean":-0.0017337799072265625,"train/train/tensor_param_model_layers_22_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_up_proj/std":0.22167971270724923,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/norm":0.0031652150292048376,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/norm":6.53125,"train/train/layer__model_layers_18/param/max_abs":1,"train/train/tensor_act_model_layers_23_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_down_proj/std":0.09143091322389986,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/max_abs":0.126953125,"train/train/layer__model_layers_76/param/max_abs":1,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/norm":8.8125,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/std":5.864707230909286e-05,"train/train/tensor_act_model_layers_5_self_attn_o_proj/mean":-0.0006551742553710938,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/norm":0.015987162769226692,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/std":0.036376953125,"train/train/tensor_act_model_layers_22_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn_k_proj/std":0.8994158511553032,"train/train/tensor_act_model_layers_27_self_attn_k_proj/norm":4931.676651967859,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/norm":0.024172233702174844,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/max_abs":0.2138671875,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/norm":0.017076650704201342,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/mean":-7.927417755126953e-06,"train/train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/std":0.00010252574670119243,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/norm":0.013785266677963271,"train/train/tensor_act_model_layers_35_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/max_abs":0.12890625,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/mean":1.3387762010097504e-09,"train/train/layer_model_layers_14/act/std":0.6905362382253967,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/max_abs":0.00092315673828125,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/mean":-6.062909960746765e-07,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/std":0.00013072094148743274,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/norm":6.78125,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/norm":0.03375569719336202,"train/train/tensor_act_model_layers_80_mlp/max_abs":1.7265625,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/max_abs":0.1904296875,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/max_abs":0.1181640625,"train/train/tensor_param_model_layers_12_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/max_abs":0.1494140625,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/norm":0.03151258771158591,"train/train/tensor_act_model_layers_92_self_attn_k_proj/max_abs":6.71875,"train/train/layer_model_layers_12/grad/max_abs":0.00092315673828125,"train/train/tensor_param_model_layers_30_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_56_mlp_down_proj/mean":-0.0044708251953125,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_57/act/max_abs":14.5625,"train/train/tensor_act_model_layers_49_mlp_down_proj/std":0.08020047800313287,"train/train/layer_model_layers_67/act/norm":14946.328607839983,"train/train/tensor_act_model_layers_5_self_attn_v_proj/std":0.35205212908948424,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_24/grad/norm":0.03610234546990375,"train/train/tensor_act_model_layers_35_input_layernorm/mean":0.0723876953125,"train/train/layer_model_layers_49/grad/max_abs":0.0010528564453125,"train/train/tensor_act_model_layers_20_post_attention_layernorm/norm":5792.611328129034,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/max_abs":0.000896453857421875,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp/norm":605.1734946376591,"train/train/tensor_act_model_layers_48_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_64/act/norm":14918.487908887184,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/mean":-5.336478352546692e-07,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4/std":1.496119262827002,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_18/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/max_abs":0.2265625,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/mean":-4.361936589702964e-09,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/max_abs":0.000782012939453125,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_41_self_attn_o_proj/norm":225.5613130784624,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/norm":0.021093188384315013,"train/train/tensor_act_model_layers_2_post_attention_layernorm/norm":5792.606079106022,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/std":4.760842070279443e-05,"train/train/tensor_act_model_layers_42_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_up_proj/max_abs":2.78125,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_v_proj/std":0.3417970144748403,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/max_abs":0.236328125,"train/train/tensor_act_model_layers_54_self_attn_k_proj/std":0.8046933580157555,"train/train/tensor_act_model_layers_55_self_attn_q_proj/mean":-0.031219482421875,"train/train/layer__model_layers_75/param/max_abs":1,"train/train/tensor_act_model_layers_79_self_attn_k_proj/mean":-0.06884765625,"train/train/tensor_act_model_layers_62_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_q_proj/mean":0.049072265625,"train/train/layer_model_layers_36/grad/std":4.9058258312631434e-05,"train/train/tensor_act_model_layers_10_self_attn_v_proj/norm":1744.3719635327138,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/std":7.20222070734205e-05,"train/train/tensor_param_model_layers_46_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_59/mean":0.063720703125,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/max_abs":0.00022125244140625,"train/train/layer_model_layers_47/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/std":0.029052734375,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/mean":0.00032806396484375,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/max_abs":0.189453125,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/mean":-7.35744833946228e-07,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/norm":3.328125,"train/train/tensor_act_model_layers_37_self_attn_k_proj/max_abs":4.5,"train/train/tensor_act_model_layers_5_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp/std":0.139161025220266,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/norm":0.010943647768576868,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/std":0.0263671875,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/std":5.6475197827191325e-05,"train/train/tensor_param_model_layers_61_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/std":4.7999358738792496e-05,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/mean":-1.3923272490501404e-07,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn_o_proj/mean":-0.0004601478576660156,"train/train/tensor_act_model_layers_37_mlp_down_proj/std":0.05822766701866182,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/std":0.03662109375,"train/train/tensor_act_model_layers_67_self_attn_k_proj/max_abs":5.09375,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_q_proj/norm":5330.847175356461,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/mean":0.0004634857177734375,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/max_abs":0.3125,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/norm":0.03552686526336491,"train/train/tensor_act_model_layers_73_self_attn_o_proj/std":0.0529795311946369,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31/mean":0.066650390625,"train/train/layer_model_layers_57/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/std":2.7931773162272636e-05,"train/train/tensor_act_model_layers_31_self_attn_v_proj/mean":0.001949310302734375,"train/train/tensor_act_model_layers_85/max_abs":11.25,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp_down_proj/norm":235.36975924923593,"train/train/tensor_param_model_layers_55_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_v_proj/mean":0.00278472900390625,"train/train/tensor_act_model_layers_52_post_attention_layernorm/std":1.0000001396983764,"train/train/tensor_act_model_layers_79_input_layernorm/mean":0.05389404296875,"train/train/tensor_act_model_layers_43_mlp_up_proj/norm":3530.6066326440828,"train/train/layer_model_layers_34/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_post_attention_layernorm/norm":5792.609008792627,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_input_layernorm/std":1.0000000372529023,"train/train/layer_model_layers_40/act/norm":14514.946209622285,"train/train/tensor_act_model_layers_52_self_attn/std":0.053658940772566284,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/mean":-0.0007781982421875,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/std":5.84214380882882e-05,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_v_proj/max_abs":2.796875,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/max_abs":0.1650390625,"train/train/tensor_param_model_layers_86_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_77/param/mean":0.00136767311512773,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/std":0.021484375,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/std":2.4659808931236055e-05,"train/train/tensor_param_model_layers_87_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/std":0.04443359375,"train/train/tensor_act_model_layers_12_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_input_layernorm/norm":5792.608154298054,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/norm":4.28125,"train/train/tensor_act_model_layers_27_mlp_up_proj/max_abs":2.265625,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/mean":-1.232139766216278e-06,"train/train/tensor_param_model_layers_10_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/std":7.948523695945962e-06,"train/train/tensor_act_model_layers_81_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_20_mlp_down_proj/std":0.03747578474855653,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_o_proj/norm":582.3355205777782,"train/train/tensor_act_model_layers_4_input_layernorm/max_abs":4.96875,"train/train/tensor_act_model_layers_28_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/max_abs":0.09912109375,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp_down_proj/std":0.04443359571498825,"train/train/layer_model_layers_63/grad/std":7.292077286819768e-05,"train/train/tensor_act_model_layers_86/std":1.9414156293738392,"train/train/tensor_param_model_layers_67_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/mean":-0.00015163421630859375,"train/train/tensor_act_model_layers_35_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_post_attention_layernorm/norm":5792.606079102153,"train/train/tensor_act_model_layers_85_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/norm":9.375,"train/train/tensor_act_model_layers_84_mlp_down_proj/norm":1347.6208786799089,"train/train/tensor_act_model_layers_34_mlp/std":0.05096452678720153,"train/train/tensor_act_model_layers_9_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/norm":0.003390765931017638,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/norm":0.009328450428894741,"train/train/tensor_act_model_layers_8/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_v_proj/max_abs":2.4375,"train/train/tensor_param_model_layers_25_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_10/frac_near_dtype_limit":0,"train/train/layer_model_layers_43/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16/norm":8033.831916081866,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/std":2.6640331733900538e-05,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp/std":0.07739296574224763,"train/train/tensor_act_model_layers_70_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/max_abs":0.10888671875,"train/train/tensor_act_model_layers_35_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/std":7.33259585992281e-05,"train/train/tensor_act_model_layers_37_self_attn/std":0.05114804200835862,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/std":0.031982421875,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/max_abs":0.1376953125,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/mean":-0.000518798828125,"train/train/tensor_act_model_layers_41/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/norm":4.75,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/norm":5.21875,"train/train/tensor_act_model_layers_73_self_attn_o_proj/mean":0.0019779205322265625,"train/train/layer_model_layers_82/grad/std":9.155035566477284e-05,"train/train/tensor_act_model_layers_7/mean":-0.0067596435546875,"train/train/layer_model_layers_57/grad/max_abs":0.000888824462890625,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/mean":-3.6670826375484467e-07,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/std":7.871249631171552e-06,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/max_abs":0.09716796875,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/mean":0.000186920166015625,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/std":0.044921875,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/mean":-0.0003795623779296875,"train/train/tensor_act_model_layers_37_mlp/std":0.05822766701866182,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/norm":4.09375,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/norm":0.03130532374353077,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/norm":0.00972151330319796,"train/train/layer_model_layers_68/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/max_abs":0.0002117156982421875,"train/train/tensor_act_model_layers_66_self_attn_v_proj/max_abs":3.5,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/mean":-9.3802809715271e-06,"train/train/layer_model_layers_3/grad/std":6.538891963916826e-05,"train/train/tensor_act_model_layers_32_self_attn_q_proj/norm":5468.988022844783,"train/train/tensor_act_model_layers_51_mlp/std":0.0767825229742154,"train/train/tensor_act_model_layers_60_self_attn_v_proj/norm":2669.4791754466214,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp/mean":-0.0015506744384765625,"train/train/tensor_act_model_layers_1_self_attn_v_proj/std":0.3320313464192643,"train/train/tensor_act_model_layers_48_mlp_up_proj/max_abs":2.96875,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/norm":7.3125,"train/train/tensor_act_model_layers_61_post_attention_layernorm/std":1.0000001098960578,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/max_abs":0.1357421875,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10/std":1.4414327425471778,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp_down_proj/norm":1567.4528736602604,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/norm":6,"train/train/layer_model_layers_93/act/max_abs":30,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp_up_proj/norm":6670.541670708159,"train/train/layer_model_layers_73/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/std":0.042724609375,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/max_abs":0.0001678466796875,"train/train/tensor_act_model_layers_44_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/std":6.313649078006753e-05,"train/train/tensor_act_model_layers_93_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_o_proj/std":0.06281032948901312,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/max_abs":0.11376953125,"train/train/tensor_act_model_layers_58_mlp/norm":558.7376441442592,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/mean":-0.0004749298095703125,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/norm":0.014501630734543206,"train/train/tensor_act_model_layers_90_mlp_up_proj/max_abs":5.8125,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/mean":0.000457763671875,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/mean":0.000553131103515625,"train/train/tensor_param_model_layers_36_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_16_mlp/std":0.03216570559168148,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/mean":2.5510787963867188e-05,"train/train/layer__model_layers_46/param/mean":0.0014039990310549923,"train/train/tensor_act_model_layers_40_self_attn_q_proj/norm":6742.159908739967,"train/train/tensor_param_model_layers_63_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_73_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp_up_proj/norm":2150.0351306802922,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/mean":-6.9588422775268555e-06,"train/train/tensor_act_model_layers_33_self_attn_o_proj/std":0.047546551212269104,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/norm":0.04185445509273684,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/mean":-1.3271346688270569e-08,"train/train/tensor_act_model_layers_64_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/norm":0.0029576280525782325,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/norm":0.021955780330274538,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/mean":-0.000148773193359375,"train/train/tensor_act_model_layers_43/std":1.294938278669713,"train/train/tensor_act_model_layers_71_self_attn/norm":454.1523125441613,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/norm":0.034448493075251344,"train/train/tensor_act_model_layers_49_self_attn/mean":-0.00323486328125,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_k_proj/std":1.0312508501560507,"train/train/tensor_act_model_layers_17_mlp/mean":0.0034942626953125,"train/train/tensor_act_model_layers_74/frac_near_user_limit":0,"train/train/layer__model_layers_44/param/max_abs":1,"train/train/tensor_act_model_layers_41_mlp_down_proj/std":0.06653403423864701,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/std":0.05126953125,"train/train/tensor_act_model_layers_27/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/mean":-3.227614797651768e-08,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/std":0.0206298828125,"train/train/tensor_act_model_layers_45_self_attn_o_proj/mean":-0.0002086162567138672,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/mean":-4.039611667394638e-08,"train/train/tensor_act_model_layers_92_self_attn_k_proj/std":0.8720724312990428,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_26/param/std":0.04890120477648718,"train/train/tensor_act_model_layers_59_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp_down_proj/std":0.2695312690043788,"train/train/tensor_param_model_layers_48_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/std":7.71485577214429e-05,"train/train/tensor_act_model_layers_87/norm":11784.779312707311,"train/train/layer__model_layers_54/param/std":0.052715725492102516,"train/train/tensor_act_model_layers_61_input_layernorm/std":1.000000067055223,"train/train/tensor_act_model_layers_83_self_attn_k_proj/norm":5369.3446268436255,"train/train/layer__model_layers_75/param/std":0.0578001364862307,"train/train/layer_model_layers_2/act/max_abs":18.5,"train/train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/norm":5.5625,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/mean":-0.000385284423828125,"train/train/tensor_param_model_layers_78_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_57_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_77/param/norm":23.444957147007116,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_59/grad/norm":0.048546672665779736,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/norm":0.01748942140599309,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/norm":0.03013659899348497,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/norm":0.008862086101600122,"train/train/tensor_act_model_layers_74_self_attn_v_proj/norm":2420.2059915658747,"train/train/tensor_act_model_layers_9_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_86/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/std":4.2880385798659976e-05,"train/train/layer__model_layers_67/param/max_abs":1,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/norm":7.59375,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/std":0.00012071827164131998,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_mlp_up_proj/norm":5408.414299094613,"train/train/tensor_param_model_layers_91_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_38/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/norm":0.0029310183681134877,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_o_proj/norm":1283.6114510215396,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/max_abs":0.1240234375,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/norm":0.019048249906214308,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/norm":0.0030189756132654733,"train/train/tensor_act_model_layers_46_mlp_down_proj/norm":449.96042114531235,"train/train/tensor_act_model_layers_5_mlp_up_proj/max_abs":3.875,"train/train/tensor_act_model_layers_65_mlp/mean":-0.002948760986328125,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/max_abs":0.00012302398681640625,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/max_abs":0.162109375,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/std":3.739471037042666e-05,"train/train/tensor_act_model_layers_72_self_attn/mean":0.002948760986328125,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/mean":-0.0003948211669921875,"train/train/tensor_act_model_layers_47_mlp_up_proj/norm":3808.7644572013364,"train/train/tensor_act_model_layers_65/max_abs":13.4375,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_35/act/norm":14009.14126087006,"train/train/tensor_act_model_layers_76_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn/norm":589.9938815628393,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/std":0.04833984375,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_87/param/norm":24.13714072751783,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model/std":1.0000003725289606,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_6/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_v_proj/max_abs":2.09375,"train/train/tensor_act_model_layers_2_mlp_up_proj/norm":2974.391936507351,"train/train/tensor_act_model_layers_83_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/norm":0.006236179538800138,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/max_abs":0.000629425048828125,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/max_abs":0.1376953125,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/norm":0.042677499138585887,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/mean":-0.000659942626953125,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/std":0.00012436443986501823,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_46_self_attn_v_proj/max_abs":2.640625,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/max_abs":0.00040435791015625,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23/norm":7841.990299231839,"train/train/tensor_act_model_layers_86_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn_v_proj/norm":2222.583950709577,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/norm":4.75,"train/train/layer_model_layers_52/grad/std":4.898490229483805e-05,"train/train/tensor_act_model_layers_70_self_attn_v_proj/mean":-0.00615692138671875,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/max_abs":0.10498046875,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_70_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/mean":0.000568389892578125,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/norm":0.013085565988761328,"train/train/layer_model_layers_70/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_input_layernorm_weight/mean":1,"train/train/layer_model_layers_78/act/mean":-0.010975049092219425,"train/train/tensor_act_model_layers_47_self_attn_o_proj/mean":-0.003467559814453125,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/std":0.049072265625,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/norm":9.0625,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/std":3.370262371130245e-05,"train/train/tensor_act_model_layers_20_self_attn/mean":0.0001571178436279297,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/mean":0.00013637542724609375,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/max_abs":0.1572265625,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/mean":-4.302710294723511e-06,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp/norm":387.6566382657851,"train/train/tensor_act_model_layers_93_input_layernorm/norm":5792.608520511222,"train/train/tensor_act_model_layers_30_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/max_abs":0.208984375,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/mean":-2.177839633077383e-07,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/max_abs":0.00016117095947265625,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/max_abs":0.0004558563232421875,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/std":7.577634167541627e-05,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/max_abs":0.1416015625,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_input_layernorm/max_abs":5.46875,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/max_abs":0.0001773834228515625,"train/train/tensor_act_model_layers_20_self_attn_o_proj/std":0.022553727131681547,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/std":4.531455991763198e-05,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/max_abs":0.1943359375,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_22_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_39/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn/max_abs":1.8671875,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/max_abs":9.34600830078125e-05,"train/train/tensor_act_model_layers_93_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_input_layernorm/max_abs":5.8125,"train/train/tensor_act_model_layers_31_mlp/max_abs":0.7890625,"train/train/tensor_act_model_layers_72_self_attn_k_proj/max_abs":5.375,"train/train/layer__model_layers_43/param/mean":0.001484857520521524,"train/train/tensor_act_model_layers_21_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp_up_proj/mean":-0.1036376953125,"train/train/layer_model_layers_52/grad/mean":-1.2616340692813237e-06,"train/train/tensor_act_model_layers_81_self_attn_v_proj/norm":2777.5890095541504,"train/train/tensor_act_model_layers_55_mlp_up_proj/std":0.40820332577349094,"train/train/tensor_act_model_layers_66/mean":0.05181884765625,"train/train/tensor_act_model_layers_16/std":1.3867463240098303,"train/train/tensor_act_model_layers_81_mlp_up_proj/max_abs":4.34375,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/mean":-6.723403930664062e-05,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_46/grad/std":5.3312060430920495e-05,"train/train/tensor_act_model_layers_76_self_attn_k_proj/norm":4844.11560332601,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/max_abs":0.000640869140625,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/mean":4.6798959374427795e-07,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/mean":0.00055694580078125,"train/train/tensor_act_model_layers_64_self_attn_k_proj/norm":4910.702723166806,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/norm":0.0007829497703757087,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_74/act/mean":0.0031343606802133415,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/max_abs":0.1279296875,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/max_abs":0.1943359375,"train/train/tensor_act_model_layers_73_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_63/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/mean":0.0004730224609375,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/max_abs":0.001190185546875,"train/train/layer_model_layers_6/act/std":0.7534980853839944,"train/train/tensor_act_model_layers_77_self_attn_k_proj/std":0.8837952250060312,"train/train/tensor_act_model_layers_56_mlp/norm":486.27301085125583,"train/train/tensor_act_model_layers_37/max_abs":17.125,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/norm":6.40625,"train/train/layer_model_layers_49/act/max_abs":15.625,"train/train/layer_model_layers_14/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_input_layernorm/mean":0.04681396484375,"train/train/tensor_act_model_layers_22_mlp_up_proj/mean":-0.0655517578125,"train/train/tensor_act_model_layers_90_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/max_abs":0.000614166259765625,"train/train/tensor_act_model_layers_69_input_layernorm/std":1.0000012572846853,"train/train/tensor_act_model_layers_42_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_post_attention_layernorm/mean":0.044189453125,"train/train/tensor_param_model_layers_88_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_7/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/mean":6.547197699546814e-07,"train/train/tensor_act_model_layers_25_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/mean":-7.963180541992188e-05,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/max_abs":0.2421875,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_up_proj/mean":-0.07421875,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/std":0.028076171875,"train/train/tensor_act_model_layers_39_self_attn_v_proj/max_abs":2.4375,"train/train/tensor_act_model_layers_15_post_attention_layernorm/mean":0.0088958740234375,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/norm":0.0006406988190070281,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_q_proj/mean":0.0408935546875,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/max_abs":0.000469207763671875,"train/train/tensor_act_model_layers_27/mean":0.05419921875,"train/train/tensor_act_model_layers_45_mlp_up_proj/std":0.3525404996129382,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/norm":4.6875,"train/train/tensor_act_/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/max_abs":0.000568389892578125,"train/train/tensor_act_model_layers_43_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_72/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp_down_proj/std":0.047119144060759836,"train/train/tensor_act_model_layers_61/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/max_abs":0.00017642974853515625,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/mean":-0.000385284423828125,"train/train/tensor_act_model_layers_39_post_attention_layernorm/std":0.9960942586261722,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/max_abs":0.000957489013671875,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/max_abs":0.12890625,"train/train/tensor_act_model_layers_51_mlp/max_abs":0.75390625,"train/train/tensor_act_model_layers_72_self_attn_k_proj/std":0.7841821079050527,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/max_abs":0.1044921875,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_mlp_down_proj/mean":-0.00455474853515625,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/mean":-0.0002079010009765625,"train/train/tensor_act_model_layers_21_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_post_attention_layernorm/max_abs":5.78125,"train/train/layer_model_layers_62/act/max_abs":13.5,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn_k_proj/mean":-0.002262115478515625,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/max_abs":0.107421875,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/norm":0.02158445111685598,"train/train/tensor_act_model_layers_15_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/std":0.0537109375,"train/train/tensor_act_model_layers_19_input_layernorm/std":1.0000000461004663,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/mean":2.8335489332675934e-07,"train/train/tensor_act_model_layers_17_mlp_down_proj/mean":0.0034942626953125,"train/train/tensor_act_model_layers_56_self_attn_q_proj/mean":-0.00743865966796875,"train/train/tensor_param_model_layers_39_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/std":5.0495838608419344e-05,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_21/param/mean":0.0016180089037057576,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/std":0.0235595703125,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/norm":0.011878043587798795,"train/train/layer__model_layers_54/param/norm":21.364037987717232,"train/train/tensor_act_model_layers_37_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/std":3.9168548784970295e-05,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/mean":-3.841705620288849e-09,"train/train/tensor_param_model_layers_53_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/mean":1.4960765838623047e-05,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13/mean":0.00626373291015625,"train/train/tensor_act_model_layers_16_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_78/param/mean":0.0014950280628412637,"train/train/tensor_act_model_layers_93_input_layernorm/std":1.0000007655468632,"train/train/tensor_act_model_layers_64_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/std":8.965138595565627e-05,"train/train/tensor_act_model_layers_9/std":1.4551030468086599,"train/train/tensor_act_model_layers_32_self_attn_o_proj/std":0.04113863411245636,"train/train/layer_model_layers_58/act/norm":13827.715450297033,"train/train/layer_model_layers_84/act/mean":-0.015866206242487982,"train/train/tensor_act_model_layers_86_self_attn_v_proj/max_abs":3,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/max_abs":0.173828125,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_51/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/max_abs":0.1259765625,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn/norm":1283.6114510215396,"train/train/tensor_act_model_layers_72_self_attn/max_abs":1.34375,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/norm":0.017805221978099778,"train/train/layer_model_layers_89/grad/std":0.00012243121820581403,"train/train/layer_model_layers_52/act/std":0.6530260325062891,"train/train/tensor_act_model_layers_79_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_post_attention_layernorm/mean":0.0635986328125,"train/train/tensor_act_model_layers_49_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/std":2.6979173730422216e-05,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/mean":5.173683166503906e-05,"train/train/tensor_param_model_layers_35_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_21_self_attn_o_proj/std":0.061401816146291824,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/mean":1.388543751090765e-07,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/norm":0.035493504628473393,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/std":4.972168514100008e-05,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/mean":1.0023359209299088e-07,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_post_attention_layernorm/max_abs":6.1875,"train/train/tensor_act_model_layers_14_self_attn_v_proj/std":0.28515627637674845,"train/train/tensor_act_model_layers_79_mlp/norm":1145.4421652145625,"train/train/layer__model_layers_60/param/norm":22.5230567627376,"train/train/tensor_act_model_layers_64_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/std":0.046630859375,"train/train/tensor_param_model_layers_68_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_input_layernorm/mean":0.060302734375,"train/train/layer__model_layers_37/param/std":0.05061769595458926,"train/train/tensor_act_model_layers_49_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/max_abs":0.12158203125,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_v_proj/norm":2329.274382508479,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/max_abs":0.00020694732666015625,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/norm":3.8125,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/std":0.040283203125,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/norm":0.0010621534587282593,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/max_abs":0.00021457672119140625,"train/train/layer_model_layers_70/act/norm":15835.506075232555,"_wandb":{"runtime":3738},"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/norm":0.029455012455777183,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/mean":-6.111804395914078e-08,"train/train/tensor_act_model_layers_26_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/norm":0.027898141077967103,"train/train/tensor_act_model_layers_54_mlp/std":0.09985381874840575,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/std":0.0220947265625,"train/train/tensor_act_model_layers_43_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_27/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/max_abs":0.00022411346435546875,"train/train/tensor_act_model_layers_9/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/max_abs":0.000301361083984375,"train/train/tensor_act_model_layers_38_mlp_down_proj/norm":353.5603615532615,"train/train/tensor_act_model_layers_77_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_5/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/mean":7.05718994140625e-05,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_k_proj/std":1.330082700108125,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/max_abs":0.000545501708984375,"train/train/tensor_act_model_layers_55_self_attn_v_proj/std":0.4101563053471664,"train/train/tensor_act_model_layers_90_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_input_layernorm/norm":5792.609497073351,"train/train/tensor_act_model_layers_65_self_attn_q_proj/mean":-0.0131072998046875,"train/train/tensor_act_model_layers_16_self_attn_k_proj/mean":-0.0082550048828125,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_post_attention_layernorm/std":1.0000024046719964,"train/train/tensor_act_model_layers_85_self_attn/mean":0.0002055540680885315,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/mean":1.0065559763461351e-07,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_post_attention_layernorm/std":0.9960939519545406,"train/train/tensor_act_model_layers_86_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/max_abs":0.1298828125,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_down_proj/mean":0.037109375,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/mean":-0.0002498626708984375,"train/train/tensor_act_model_layers_83_self_attn_v_proj/max_abs":2.921875,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/std":0.00011613682909862579,"train/train/tensor_act_model_layers_43_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/max_abs":8.821487426757812e-05,"train/train/tensor_act_model_layers_82_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_24/grad/std":4.4598255463954e-05,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/std":0.04248046875,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/mean":-8.821487426757812e-05,"train/train/tensor_act_model_layers_25_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/std":0.0400390625,"train/train/tensor_act_model_layers_54_self_attn_o_proj/std":0.05786356790851186,"train/train/layer__model_layers_31/param/max_abs":1,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/std":4.8416628754745994e-05,"train/train/tensor_act_model_layers_26/std":1.3261877235960011,"train/train/tensor_act_model_layers_76_self_attn_v_proj/mean":0.0020503997802734375,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/std":0.057373046875,"train/train/tensor_act_model_layers_93_input_layernorm/max_abs":5.3125,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/mean":0.0002079010009765625,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/mean":-1.4132820069789886e-07,"train/train/tensor_param_model_layers_80_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/std":0.036376953125,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/mean":1.5506520867347717e-06,"train/train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_v_proj/mean":0.0104217529296875,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/mean":1.1861324310302734e-05,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/norm":0.001183214312107274,"train/train/tensor_act_model_layers_54_self_attn_o_proj/norm":335.54180458061114,"train/train/tensor_act_model_layers_83_mlp_down_proj/std":0.23144531853591332,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/norm":6.28125,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_25_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_q_proj/norm":5113.568230649768,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_post_attention_layernorm/std":0.9960938696767698,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_up_proj/norm":2964.256895347004,"train/train/tensor_act_model_layers_8_post_attention_layernorm/norm":5792.60925293498,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn/std":0.11914136339910524,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/max_abs":0.0006866455078125,"train/train/tensor_act_model_layers_59_input_layernorm/mean":0.03765869140625,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/norm":0.019749660000107794,"train/train/tensor_act_model_layers_19_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_up_proj/max_abs":2.53125,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/mean":4.4345855712890625e-05,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/norm":0.02170012258437906,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/mean":-1.628883183002472e-06,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/mean":0.00018978118896484375,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_mlp_down_proj/std":0.23291070481201764,"train/train/tensor_act_model_layers_45_self_attn_k_proj/mean":0.0728759765625,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_67_mlp_down_proj/mean":0.00476837158203125,"train/train/tensor_act_model_layers_82_mlp_up_proj/max_abs":4.59375,"train/train/tensor_act_model_layers_54_post_attention_layernorm/mean":0.0458984375,"train/train/layer__model_layers_30/param/norm":20.332688284893,"train/train/tensor_act_model_layers_92/max_abs":17.75,"train/train/tensor_act_model_layers_51_mlp_down_proj/std":0.0767825229742154,"train/train/tensor_act_model_layers_75_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/norm":0.0008395844505585513,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/std":2.1760209983479737e-05,"train/train/layer_model_layers_43/grad/norm":0.04383669811163795,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/mean":-1.2415694072842598e-07,"train/train/tensor_act_model_layers_30/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_mlp_down_proj/norm":1344.3798457180258,"train/train/tensor_act_model_layers_57_self_attn_o_proj/std":0.095337186747636,"train/train/tensor_act_model_layers_29_post_attention_layernorm/norm":5792.6141357450415,"train/train/tensor_act_model_layers_9_input_layernorm/max_abs":5.6875,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/norm":0.01782247464828197,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/norm":0.0009609270871817473,"train/train/layer_model_layers_1/act/norm":16850.732498073685,"train/train/tensor_act_model_layers_70_self_attn_k_proj/norm":5293.177467122905,"train/train/tensor_act_model_layers_23_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/std":0.00011016182021338669,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_q_proj/max_abs":5.84375,"train/train/tensor_act_model_layers_75_self_attn/max_abs":3.171875,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22/mean":0.03753662109375,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_mlp_down_proj/std":0.8183711012407564,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/max_abs":7.104873657226562e-05,"train/train/tensor_act_model_layers_58_post_attention_layernorm/max_abs":5.6875,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/std":3.194413257680392e-05,"train/train/tensor_act_model_layers_14_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/mean":-0.000598907470703125,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/max_abs":0.1171875,"train/train/tensor_act_model_layers_60_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_v_proj/std":0.41210940798029383,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/max_abs":0.00018596649169921875,"train/train/tensor_act_model_layers_15/max_abs":18.125,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/max_abs":0.00054931640625,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/max_abs":0.0001316070556640625,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp_up_proj/std":0.267578125,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/mean":0.00011730194091796875,"train/train/layer_model_layers_9/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_82/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_up_proj/std":0.27832202576227066,"train/train/tensor_act_model_layers_81/std":1.740242235570552,"train/train/tensor_act_model_layers_39_self_attn_k_proj/std":0.8271549409252618,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/mean":-0.000347137451171875,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/mean":-1.4761462807655334e-07,"train/train/layer_model_layers_73/act/mean":-0.005906105041503906,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/mean":3.026798367500305e-09,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/mean":-0.00026702880859375,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/norm":5.625,"train/train/tensor_param_model_layers_23_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_10_self_attn_o_proj/std":0.06774949313121226,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/std":1.2073944908728782e-05,"train/train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/norm":0.0007586741427930851,"train/train/tensor_act_model_layers_7_self_attn_q_proj/std":1.5234378716884063,"train/train/layer_model_layers_34/grad/std":5.005656593817627e-05,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/mean":7.635913789272308e-06,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_77/grad/mean":2.2342791795358643e-06,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/max_abs":0.00057220458984375,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/mean":3.029126673936844e-07,"train/train/tensor_act_model_layers_73_self_attn_k_proj/mean":0.019317626953125,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/mean":-1.990795135498047e-05,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/max_abs":0.12890625,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/std":4.154817608029496e-05,"train/train/layer_model_layers_23/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn/mean":0.00139617919921875,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/std":0.050048828125,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/std":0.039306640625,"train/train/tensor_act_model_layers_84_mlp_up_proj/std":0.6015626486245504,"train/train/layer_model_layers_50/grad/max_abs":0.0009002685546875,"train/train/layer_model_layers_79/grad/norm":0.06946386139143218,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/std":4.1239529676656264e-05,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/max_abs":0.125,"train/train/tensor_act_model_layers_52_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/std":1.803573495956883e-05,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/norm":0.018408652383598994,"train/train/tensor_act_model_layers_86_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/std":5.683738391473275e-05,"train/train/tensor_act_model_layers_58_self_attn_o_proj/norm":493.29418472058626,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_post_attention_layernorm/max_abs":6.0625,"train/train/tensor_act_model_layers_36_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/std":0.0228271484375,"train/train/tensor_act_model_layers_88/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_input_layernorm/max_abs":5.46875,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/mean":4.306435585021973e-06,"train/train/tensor_act_model_layers_41_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/mean":0,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/std":3.4114031550388115e-05,"train/train/tensor_act_model_layers_28_mlp/std":0.04895035049242646,"train/train/tensor_act_model_layers_21_input_layernorm/std":1.0000000447034827,"train/train/tensor_act_model_layers_70_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/std":3.12976578471559e-05,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_down_proj/std":0.04400715317961084,"train/train/layer_model_layers_9/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_38_mlp/std":0.060302740166544154,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/norm":5.3125,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/mean":-8.843839168548584e-06,"train/train/tensor_act_model_layers_47_self_attn_k_proj/mean":0.010223388671875,"train/train/tensor_act_model_layers_33_input_layernorm/std":0.9960938322777808,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/std":5.799177567988558e-05,"train/train/tensor_act_model_layers_68_input_layernorm/norm":5792.6062011744325,"train/train/tensor_act_model_layers_67_self_attn_o_proj/norm":551.3289695599515,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/norm":0.0003916177391042977,"train/train/tensor_act_model_layers_37_self_attn_v_proj/std":0.3300781368812511,"train/train/tensor_act_model_layers_51_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_v_proj/mean":0.00785064697265625,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43/mean":0.0721435546875,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/mean":3.795139491558075e-08,"train/train/tensor_act_model_layers_38_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/max_abs":0.000682830810546875,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/mean":-0.00115203857421875,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/mean":-6.031990051269531e-05,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/max_abs":0.134765625,"train/train/layer_model_layers_20/act/mean":0.004336100358229417,"train/train/tensor_act_model_layers_69_mlp_up_proj/std":0.49609381007396913,"train/train/tensor_act_model_layers_58_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_o_proj/norm":355.72957544015867,"train/train/tensor_act_model_layers_1_input_layernorm/norm":5792.618896485132,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29/norm":7663.682615986749,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/std":0.0289306640625,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/max_abs":0.1357421875,"train/train/tensor_act_model_layers_26_post_attention_layernorm/max_abs":6.15625,"train/train/tensor_act_model_layers_5_post_attention_layernorm/max_abs":5.15625,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/mean":-1.5974044799804688e-05,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/norm":6.46875,"train/train/tensor_act_model_layers_42_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/std":0.03955078125,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/std":0.054443359375,"train/train/tensor_act_model_layers_16_self_attn_q_proj/std":0.9091812834433737,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/norm":0.0007270927068049531,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/max_abs":0.00037384033203125,"train/train/layer_model_layers_71/grad/mean":1.3166562001716328e-06,"train/train/tensor_act_model_layers_40/frac_near_dtype_limit":0,"train/train/tensor_act_lm_head/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn/norm":406.59723534154654,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/norm":2.796875,"train/train/tensor_act_model_layers_57_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/max_abs":0.00022125244140625,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/norm":5.96875,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/max_abs":0.1279296875,"train/train/tensor_act_model_layers_4_self_attn_q_proj/std":1.1113333609781126,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/max_abs":0.1396484375,"train/train/tensor_act_model_layers_26_self_attn/std":0.048523078549410646,"train/train/layer__model_layers_73/param/mean":0.001391358754936135,"train/train/tensor_act_model_layers_58_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_76_post_attention_layernorm/std":1.000002652403175,"train/train/tensor_act_model_layers_85_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_q_proj/mean":-0.031707763671875,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/norm":0.022050338754192694,"train/train/tensor_act_model_layers_58_input_layernorm/mean":0.0406494140625,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/std":6.763616035948802e-05,"train/train/tensor_act_model_layers_15/norm":8072.654066033473,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/mean":-0.000255584716796875,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/std":0.037109375,"train/train/layer__model_layers_60/param/mean":0.0014709535515438563,"train/train/tensor_act_model_layers_23_self_attn_k_proj/std":0.9091812834433737,"train/train/tensor_act_model_layers_68_mlp_up_proj/mean":-0.091552734375,"train/train/tensor_act_model_layers_76_post_attention_layernorm/norm":5792.605834962166,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/std":4.2657945427951545e-05,"train/train/tensor_act_model_layers_85_mlp/max_abs":2.90625,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27/std":1.3281365450188987,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/norm":0.0015355660974943762,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/std":4.55718351855496e-05,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/norm":3.671875,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/mean":6.3087791204452515e-06,"train/train/layer__model_layers_32/param/norm":20.027843167778325,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/mean":-2.637505531311035e-06,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/std":9.027095989818337e-05,"train/train/tensor_act_model_layers_37_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_input_layernorm/norm":5792.606933594368,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/max_abs":0.0009307861328125,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_22/param/max_abs":1,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_q_proj/std":0.9189474549709402,"train/train/tensor_act_model_layers_31_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/max_abs":0.0011444091796875,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/std":4.001731738534363e-05,"train/train/layer__model_layers_35/param/max_abs":1,"train/train/layer_model_layers_14/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/std":3.1392997161657554e-05,"train/train/tensor_act_model_layers_92_mlp/norm":4739.755600762709,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/mean":0.000278472900390625,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn/mean":-0.00037097930908203125,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76/norm":9239.83660841937,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/mean":9.145587682723999e-07,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/norm":5.96875,"train/train/tensor_act_model_layers_58_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_q_proj/mean":0.195556640625,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/std":1.6570893987718762e-05,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/norm":5.21875,"train/train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/std":3.022440121834922e-05,"train/train/layer_model_layers_9/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_k_proj/std":0.8505879346296338,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/mean":-1.3727694749832153e-06,"train/train/tensor_act_model_layers_87_self_attn/max_abs":3.203125,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/max_abs":0.150390625,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/norm":5.96875,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/mean":2.6935595087707043e-08,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/mean":-0.00148773193359375,"train/train/tensor_act_model_layers_57_self_attn_o_proj/mean":-0.001384735107421875,"train/train/tensor_act_model_layers_63/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_v_proj/norm":1693.1257695758375,"train/train/tensor_act_model_layers_16_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/std":0.0308837890625,"train/train/tensor_act_model_layers_66_self_attn_o_proj/std":0.20581317208535532,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/std":5.2900345273554e-05,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/mean":-1.4544639270752668e-07,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/std":1.0105734373811462e-05,"train/train/tensor_act_model_layers_13_self_attn_q_proj/norm":6206.8860386145925,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/mean":4.44706529378891e-08,"train/train/tensor_act_model_layers_15_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/std":0.03271484375,"train/train/tensor_act_model_layers_65_self_attn/std":0.1251256011552929,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/std":0.0380859375,"train/train/tensor_act_model_layers_91_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_post_attention_layernorm/mean":0.0357666015625,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/mean":-0.00014019012451171875,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/max_abs":0.000293731689453125,"train/train/layer_model_layers_81/act/norm":17360.167171817855,"train/train/tensor_act_model_layers_89_post_attention_layernorm/mean":0.069091796875,"train/train/tensor_act_model_layers_31_self_attn_k_proj/norm":4613.394073952423,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn_o_proj/norm":931.3154026047223,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62/max_abs":13.5,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_v_proj/mean":-0.0006771087646484375,"train/train/tensor_act_model_layers_63_self_attn_v_proj/std":0.4409188046663964,"train/train/tensor_act_model_layers_92/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_o_proj/std":0.16162248252325284,"train/train/tensor_act_model_layers_13_self_attn_o_proj/max_abs":0.462890625,"train/train/tensor_act_model_layers_22_post_attention_layernorm/max_abs":6.09375,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/norm":0.003820163624930457,"train/train/tensor_act_model_layers_85_self_attn_k_proj/norm":5013.891728556714,"train/train/tensor_act_model_layers_5_self_attn_k_proj/norm":7768.343749166221,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/std":0.05322265625,"train/train/layer__model_layers_37/param/max_abs":1,"train/train/tensor_act_model_layers_49_mlp/max_abs":0.90234375,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/std":0.041259765625,"train/train/layer_model_layers_80/act/norm":17346.183775680223,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_29/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/std":0.0284423828125,"train/train/layer__model_layers_14/param/max_abs":1,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/mean":0.000545501708984375,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_24_mlp_up_proj/std":0.25830248307531,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/mean":1.3248063623905182e-07,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/mean":2.2239983081817627e-06,"train/train/tensor_act_model_layers_15_self_attn_k_proj/std":1.0625000525923323,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/norm":7.03125,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_56/grad/std":5.069186771802898e-05,"train/train/layer_model_layers_45/act/frac_near_user_limit":0,"train/train/layer_model_layers_28/act/std":0.6855969800290327,"train/train/tensor_act_model_layers_39_input_layernorm/max_abs":6.0625,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/max_abs":0.1640625,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/std":3.435092711449107e-05,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/mean":-2.223532646894455e-06,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_92/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/max_abs":0.1904296875,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/mean":-3.952300176024437e-08,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/norm":0.00205247729422042,"train/train/tensor_param_model_layers_87_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_36/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/mean":-0.00013256072998046875,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_mlp/mean":-0.003368377685546875,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/max_abs":0.208984375,"train/train/layer__model_layers_32/param/mean":0.001655566710949688,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/std":8.453053866487833e-05,"train/train/layer_model_layers_44/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/max_abs":0.00078582763671875,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_42/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/max_abs":0.1591796875,"train/train/tensor_act_model_layers_64_input_layernorm/std":1.0000000670552232,"train/train/tensor_act_model_layers_23_post_attention_layernorm/std":1.0000000074505806,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/std":0.00011369660778254158,"train/train/tensor_act_model_layers_20_self_attn_v_proj/max_abs":1.8046875,"train/train/tensor_act_model_layers_49/norm":7514.195213352531,"train/train/tensor_act_model_layers_9_post_attention_layernorm/norm":5792.608886721579,"train/train/tensor_param_model_layers_43_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_86_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/norm":3.8125,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/max_abs":0.1728515625,"train/train/layer_model_layers_14/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_24_mlp/std":0.04852337578370075,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/norm":0.00437682611808025,"train/train/layer__model_layers_11/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/std":0.031005859375,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_input_layernorm/max_abs":6.28125,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/max_abs":0.11328125,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/max_abs":0.162109375,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/std":0.03955078125,"train/global_step":1500,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/norm":0.02319865919094461,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_66/act/std":0.7279626859814318,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/mean":-0.00011444091796875,"train/train/tensor_param_model_layers_86_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/mean":4.2244791984558105e-06,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/std":0.00012228664759007668,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/max_abs":0.000217437744140625,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/std":4.0653799919387854e-05,"train/train/tensor_act_model_layers_63_input_layernorm/mean":0.0589599609375,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/std":4.7098728948266605e-05,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/norm":10.8125,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp_down_proj/mean":0.00466156005859375,"train/train/tensor_act_model_layers_46_input_layernorm/max_abs":5.84375,"train/train/tensor_act_model_layers_58_self_attn_v_proj/std":0.369630025804074,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/max_abs":0.1376953125,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/std":0.040283203125,"train/train/layer_model_layers_82/act/std":0.7859590314049352,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/max_abs":0.000331878662109375,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/max_abs":0.0002613067626953125,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/mean":-4.3392181396484375e-05,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/max_abs":0.00013446807861328125,"train/train/tensor_act_model_layers_8_input_layernorm/std":1.0000000098661983,"train/train/tensor_param_model_embed_tokens_weight/mean":-0.0081787109375,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/max_abs":0.000156402587890625,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/std":7.416126644399432e-05,"train/train/tensor_act_model_layers_35/max_abs":17.625,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/norm":0.00037643568110599067,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/mean":-6.012618541717529e-06,"train/train/tensor_act_model_layers_92/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_post_attention_layernorm/mean":0.04815673828125,"train/train/layer_model_layers_8/grad/max_abs":0.00201416015625,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/mean":-0.000518798828125,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/mean":-0.00010633468627929688,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/norm":5.15625,"train/train/tensor_act_model_layers_51_mlp_down_proj/norm":444.3641786070455,"train/train/tensor_act_model_layers_88_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/std":1.0000000189224918,"train/train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_16/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp/norm":251.49561028282218,"train/train/tensor_act_model_layers_71_self_attn_o_proj/mean":-0.00051116943359375,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/norm":7.15625,"train/train/layer__model_layers_90/param/mean":0.0016386475466342873,"train/train/tensor_act_model_layers_89_self_attn/norm":1783.3119747617445,"train/train/layer__model_layers_48/param/mean":0.0013347988753534517,"train/train/tensor_param_model_layers_75_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp_up_proj/max_abs":3.34375,"train/train/layer_model_layers_17/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/max_abs":0.00139617919921875,"train/train/tensor_act_model_layers_47_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/std":1.1449369171660879e-05,"train/train/tensor_act_model_layers_59_self_attn_o_proj/mean":-0.00145721435546875,"train/train/tensor_act_model_layers_63/norm":7966.085016452143,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_58_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn_k_proj/norm":4872.321789969411,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/std":5.723405360240864e-05,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/max_abs":8.440017700195312e-05,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_v_proj/mean":0.0015954971313476562,"train/train/tensor_act_model_layers_64_self_attn_q_proj/max_abs":6.46875,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_78_self_attn_v_proj/mean":-0.004791259765625,"train/train/tensor_act_model_layers_86_input_layernorm/max_abs":5.125,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_75/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/norm":4.53125,"train/train/tensor_act_model_layers_56_mlp_up_proj/mean":-0.1016845703125,"train/train/tensor_act_model_layers_63_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/mean":-0.000244140625,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_k_proj/std":0.9726562883001726,"train/train/tensor_act_model_layers_56_self_attn_o_proj/max_abs":1.2421875,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/mean":0.00064849853515625,"train/train/tensor_act_model_layers_84_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn/mean":-0.00018164515495300293,"train/train/tensor_act_model_layers_54_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/std":0.03173828125,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/norm":2.921875,"train/train/tensor_act_model_layers_45/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_k_proj/std":0.8369157895723592,"train/train/tensor_act_model_layers_79_self_attn_o_proj/mean":0.003864288330078125,"train_loss":11.667699412027995,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/max_abs":0.2421875,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/mean":-5.0961971282958984e-05,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_q_proj/norm":5074.190265915269,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/mean":0.000171661376953125,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/norm":0.004242135462081713,"train/train/layer_model_layers_33/act/mean":-0.011575258695162259,"train/train/tensor_act_model_layers_29_mlp_down_proj/std":0.07251018947901292,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/std":4.104298819157804e-05,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/norm":0.0020690785747994583,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/mean":0.000255584716796875,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60/std":1.3398495262510834,"train/train/tensor_act_model_layers_87_self_attn_q_proj/std":1.12500008662391,"train/train/tensor_act_model_layers_6_self_attn_q_proj/std":1.3593751507243808,"train/train/tensor_act_model_layers_1_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/norm":14.1875,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/std":0.00010694919508300433,"train/train/tensor_act_model_layers_37_self_attn/norm":296.552332604813,"train/train/tensor_act_model_layers_89_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_53/grad/norm":0.04177020690666522,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/act/std":0.6599148075628809,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/norm":2.8125,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/norm":0.0024740421966928084,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/mean":0.0003509521484375,"train/train/tensor_param_model_layers_24_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_88/norm":12234.966080952045,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp_down_proj/max_abs":1.2265625,"train/train/tensor_act_model_layers_33_self_attn_k_proj/std":0.8085952841702709,"train/train/tensor_act_model_layers_67/max_abs":13.3125,"train/train/tensor_act_model_layers_42_self_attn_v_proj/max_abs":2.5625,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/max_abs":0.00150299072265625,"train/train/tensor_act_model_layers_39_self_attn/norm":359.54123512118264,"train/train/layer__model_layers_3/param/max_abs":1,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/max_abs":0.251953125,"train/train/layer_model_layers_2/grad/max_abs":0.00194549560546875,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/mean":2.3186206817626953e-05,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/norm":0.0011126171293674835,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_49/param/std":0.05302170757379157,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/mean":-1.1647352948784828e-07,"train/train/tensor_param_model_layers_56_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/mean":4.330649971961975e-08,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_o_proj/mean":0.0002644062042236328,"train/train/tensor_act_model_layers_8_self_attn_q_proj/max_abs":6.25,"train/train/tensor_act_model_layers_69/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/max_abs":0.00078582763671875,"train/train/layer__model_layers_58/param/max_abs":1,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/norm":0.0018094295806220973,"train/train/tensor_act_model_layers_82_post_attention_layernorm/norm":5792.609252936493,"train/train/tensor_act_model_layers_41_mlp_up_proj/std":0.31640632064253404,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/norm":6.9375,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/max_abs":9.34600830078125e-05,"train/train/tensor_act_model_layers_22_self_attn_k_proj/std":0.9804687803959937,"train/train/tensor_act_model_layers_10_input_layernorm/max_abs":5.6875,"train/train/tensor_act_model_layers_12_self_attn_o_proj/norm":152.75751545221004,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp_up_proj/norm":2884.245649211661,"train/train/layer_model_layers_60/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/norm":0.001085916410471224,"train/train/tensor_param_model_layers_29_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_30_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_10_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_v_proj/max_abs":2.359375,"train/train/tensor_act_model_layers_92_self_attn_o_proj/std":0.21241145607705472,"train/train/layer_model_layers_12/act/std":0.6576119623213966,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_81/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_post_attention_layernorm/mean":0.05584716796875,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/std":8.487937481790748e-05,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/std":4.508195450276296e-05,"train/train/tensor_param_model_layers_45_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/mean":2.1583400666713715e-07,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/max_abs":0.00110626220703125,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/norm":0.016572702792082575,"train/train/tensor_act_model_layers_33_self_attn_v_proj/norm":1725.9536977201797,"train/train/tensor_act_model_layers_58/std":1.324224695681837,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_71/act/std":0.7033028045800568,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/norm":0.002226881602493239,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/norm":0.001287860101868044,"train/train/tensor_act_model_layers_81_self_attn_o_proj/std":0.2172879935130244,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/max_abs":0.1318359375,"train/train/tensor_act_model_layers_57_post_attention_layernorm/max_abs":5.75,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/mean":1.2498348951339722e-06,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/max_abs":0.00021076202392578125,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/norm":3.265625,"train/train/tensor_act_model_layers_78_post_attention_layernorm/std":1.0000025238809924,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/norm":4.75,"train/train/layer__model_layers_84/param/mean":0.0013939534632166537,"train/train/tensor_act_model_layers_29_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/mean":1.1618249118328094e-07,"train/train/tensor_act_model_layers_58_self_attn/norm":493.29418472058626,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn/norm":146.8135230071949,"train/train/layer__model_layers_93/param/norm":26.09592056840494,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/norm":0.03327373401861725,"train/train/tensor_act_model_layers_39_self_attn_v_proj/norm":1876.0092205761687,"train/train/layer__model_layers_56/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn_o_proj/max_abs":3.171875,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/std":0.0220947265625,"train/train/tensor_act_model_layers_87_self_attn/norm":1587.689742085104,"train/train/tensor_act_model_layers_22_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/std":0.02392578125,"train/train/tensor_param_model_layers_85_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/max_abs":0.000904083251953125,"train/train/layer_model_layers_5/grad/norm":0.04825798505203583,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/max_abs":0.0020751953125,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/std":0.044189453125,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/max_abs":0.0011138916015625,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/std":6.20173389973659e-05,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/mean":-8.288770914077759e-08,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/norm":0.016015798288663907,"train/train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_mlp/norm":899.3549603284915,"train/train/tensor_act_model_layers_58_self_attn_q_proj/norm":4793.75873668403,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_v_proj/mean":0.0039215087890625,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_9/param/norm":20.09390795039009,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/norm":7.03125,"train/train/tensor_act_model_layers_9_mlp_down_proj/max_abs":0.318359375,"train/train/layer_model_layers_67/act/frac_near_user_limit":0,"train/train/layer__model_layers_67/param/norm":22.658168022260757,"train/train/tensor_act_model_layers_42_mlp/mean":0.0007123947143554688,"train/train/tensor_param_model_layers_50_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_57/act/mean":-0.014438335712139424,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_42/param/std":0.05136092490839355,"train/train/layer__model_layers_72/param/norm":22.91517987601232,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_74/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_k_proj/std":1.3378949186158722,"train/train/tensor_act_model_layers_67_input_layernorm/mean":0.03790283203125,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/max_abs":0.22265625,"train/train/tensor_act_model_layers_23_self_attn_v_proj/norm":2032.7828792564003,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/max_abs":0.1396484375,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/max_abs":0.00019073486328125,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_v_proj/std":0.3769531571185637,"train/train/tensor_act_model_layers_82_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn_o_proj/max_abs":1.0625,"train/train/tensor_act_model_layers_10_self_attn_k_proj/max_abs":5.15625,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/mean":-1.7249112715944648e-07,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/norm":4.9375,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/mean":-2.1886080503463745e-06,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_k_proj/max_abs":5.5,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/mean":-6.012618541717529e-06,"train/train/tensor_act_model_layers_42_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80/frac_near_user_limit":0,"train/train/layer__model_layers_40/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/mean":3.170967102050781e-05,"train/train/tensor_act_model_layers_67_self_attn_q_proj/norm":5491.756000345263,"train/train/tensor_act_model_layers_59_self_attn/std":0.10380033373155731,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_7/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/norm":7.15625,"train/train/tensor_act_model_layers_40_mlp_up_proj/norm":3389.52443296848,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/max_abs":0.00012159347534179688,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/mean":-3.094971179962158e-05,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/norm":6.28125,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/mean":-3.8176774978637695e-05,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/std":0.0390625,"train/train/tensor_act_model_layers_0_mlp/mean":0.0394287109375,"train/train/tensor_act_model_layers_31_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_8_self_attn_q_proj/norm":8375.838651176151,"train/train/tensor_act_model_layers_45_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_input_layernorm_weight/std":0,"train/train/layer_model_layers_65/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_mlp_up_proj/std":0.36718756087282367,"train/train/layer__model_layers_46/param/norm":21.183149673221994,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/max_abs":0.228515625,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_5_mlp_down_proj/std":0.05664062808299879,"train/train/tensor_param_model_layers_42_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_self_attn_k_proj/std":0.7226562809299771,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn_k_proj/mean":-0.034912109375,"train/train/tensor_act_model_layers_92_post_attention_layernorm/mean":0.05035400390625,"train/train/tensor_act_model_layers_83_mlp/mean":0.017547607421875,"train/train/layer_model_layers_71/act/max_abs":13.25,"train/train/tensor_act_model_layers_51_input_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_69/param/mean":0.0012506017818837754,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/std":6.962088753642057e-05,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/norm":6.65625,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_q_proj/norm":5874.145959149961,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/std":0.031982421875,"train/train/tensor_act_model_layers_21_input_layernorm/norm":5792.60681152909,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/norm":0.014386246573255282,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/mean":9.313225746154785e-09,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/max_abs":0.212890625,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/std":0.02392578125,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn/std":0.02526958925322281,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/mean":1.737498678267002e-08,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/norm":0.015238206332721013,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/frac_near_dtype_limit":0,"eval/runtime":16.2898,"train/train/tensor_act_model_layers_68_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/max_abs":0.0002689361572265625,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/norm":0.0009456410622725652,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn_o_proj/max_abs":1.796875,"train/train/tensor_act_model_layers_60_mlp/norm":629.2381183575887,"train/train/tensor_act_model_layers_12_mlp_down_proj/max_abs":0.4453125,"train/train/tensor_act_model_layers_86_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_input_layernorm/std":1.0000000353902572,"train/train/tensor_act_model_layers_22/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_k_proj/norm":4703.795104422978,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/mean":-6.810296326875687e-09,"train/train/tensor_act_model_layers_56_self_attn_q_proj/max_abs":6.8125,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/std":0.00011447260025263142,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/norm":6.4375,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/max_abs":0.0008392333984375,"train/train/tensor_act_model_layers_80_post_attention_layernorm/max_abs":5.6875,"train/train/tensor_act_model_layers_85/mean":0.146240234375,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_self_attn_k_proj/norm":4596.676498511059,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp/norm":316.9139937447785,"train/train/tensor_act_model_layers_60_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/std":0.05810546875,"train/train/tensor_act_model_layers_90_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_85/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/max_abs":0.0005645751953125,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/max_abs":0.1826171875,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_8/grad/std":4.929732854360831e-05,"train/train/tensor_act_model_layers_5_mlp_up_proj/std":0.24047891064916396,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/max_abs":0.1103515625,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_post_attention_layernorm/mean":0.013458251953125,"train/train/tensor_act_model_layers_20_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_17/grad/max_abs":0.001708984375,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/max_abs":0.0004405975341796875,"train/train/tensor_act_model_layers_83_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/norm":6.25,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68/norm":8468.258237655606,"train/train/tensor_param_model_layers_57_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/mean":0.00624847412109375,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/norm":0.009326703121373325,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/norm":0.02930463052457815,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/std":0.038818359375,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/max_abs":0.0010528564453125,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/std":9.210265501081385e-05,"train/train/layer_model_layers_71/grad/norm":0.05501604599080929,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/std":1.982047369389197e-05,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/std":9.557495725978198e-05,"train/train/tensor_act_model_layers_41_mlp/norm":385.94885037260656,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/mean":-0.00024127960205078125,"train/train/layer_model_layers_1/act/max_abs":18.375,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/mean":-8.285045623779297e-06,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_73/act/max_abs":13.125,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/max_abs":0.00010347366333007812,"train/train/tensor_act_model_layers_72_self_attn_o_proj/std":0.07739773695129841,"train/train/layer_model_layers_48/act/std":0.6772475823375154,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_act_model_layers_80_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/std":0.05615234375,"train/train/tensor_act_model_layers_26_self_attn_q_proj/std":0.9209005869544395,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/max_abs":0.000255584716796875,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/max_abs":0.138671875,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/std":1.5380561672612007e-05,"train/train/tensor_act_model_layers_81_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/std":0.0267333984375,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/max_abs":0.185546875,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/std":7.368773041547158e-05,"train/train/tensor_act_model_layers_24_self_attn_v_proj/mean":0.003070831298828125,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/mean":-0.000732421875,"train/train/tensor_act_model_layers_7_input_layernorm/norm":5792.6107177745425,"train/train/tensor_act_model_layers_59_post_attention_layernorm/std":1.000000104308123,"train/train/tensor_param_model_layers_21_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_82/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/max_abs":0.10302734375,"train/train/tensor_act_model_layers_61_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_69_self_attn_v_proj/max_abs":3.046875,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/max_abs":0.0002536773681640625,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/max_abs":0.00054168701171875,"train/train/tensor_act_model_layers_19_mlp_up_proj/norm":2256.29330727737,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/std":1.350455746191178e-05,"train/train/tensor_act_model_layers_75_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_63/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86/max_abs":11.8125,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/max_abs":0.000972747802734375,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/max_abs":0.0016021728515625,"train/train/layer_model_layers_13/act/norm":14459.52012119192,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/mean":-0.0003147125244140625,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/mean":-0.000823974609375,"train/train/tensor_act_model_layers_67/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_post_attention_layernorm/std":1.0000000670552232,"train/train/tensor_act_model_layers_83_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/max_abs":0.0002689361572265625,"train/train/tensor_act_model_layers_77_self_attn_q_proj/norm":5688.499710630513,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/norm":0.02350875830276918,"train/train/tensor_act_model_layers_13_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_o_proj/norm":214.72004332620855,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/norm":6.5625,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/mean":4.5099295675754547e-07,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/std":3.461786411922926e-05,"train/train/tensor_act_model_layers_15_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/mean":2.8133392333984375e-05,"train/train/tensor_act_model_layers_71_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_post_attention_layernorm/mean":0.025634765625,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/std":0.031494140625,"train/train/layer_model_layers_27/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/max_abs":0.00013446807861328125,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn_q_proj/max_abs":6.03125,"train/train/layer_model_layers_21/grad/std":4.985008075641213e-05,"train/train/tensor_act_model_layers_37_self_attn_q_proj/mean":0.052978515625,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_89/std":2.175796847741259,"train/train/tensor_act_model_layers_89_self_attn_v_proj/norm":3099.602897514928,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/mean":-8.195638656616211e-08,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/std":0.0224609375,"train/train/tensor_param_model_layers_66_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/mean":9.583309292793274e-07,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_25/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/mean":0.0003070831298828125,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_8/act/mean":-0.0011056753305288462,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/std":6.510800269142152e-05,"train/train/tensor_act_model_layers_72_self_attn_v_proj/norm":2319.187900325529,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/mean":-0.000225067138671875,"train/train/tensor_act_model_layers_17_self_attn_q_proj/std":1.0625000613577207,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/norm":0.013887476518132146,"train/train/layer_model_layers_50/act/mean":-0.005319356918334961,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/std":0.05517578125,"train/train/tensor_act_model_layers_36_input_layernorm/std":0.9960939519545406,"train/train/tensor_act_model_layers_74_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_1/grad/mean":-4.316098964809442e-07,"train/train/tensor_act_model_layers_88_self_attn_q_proj/std":1.2168016142775822,"train/train/tensor_act_model_layers_0_post_attention_layernorm/mean":0.0347900390625,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/std":3.07217173726617e-05,"train/train/tensor_act_model_rotary_emb/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_o_proj/std":0.1606485554953938,"train/train/tensor_act_model_layers_70_self_attn/norm":1506.1802860033656,"train/train/tensor_act_model_norm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/std":0.045166015625,"train/train/tensor_act_model_layers_36_self_attn_q_proj/std":0.9736345739740873,"train/train/tensor_act_model_layers_56_self_attn_v_proj/std":0.3417968944140838,"train/train/tensor_act_model_layers_34_post_attention_layernorm/max_abs":6.25,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_75/grad/max_abs":0.00165557861328125,"train/train/tensor_act_model_layers_52_self_attn_q_proj/max_abs":5.28125,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/std":6.952782664087386e-05,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_mlp_up_proj/mean":-0.0994873046875,"train/train/tensor_act_model_layers_58_mlp_up_proj/norm":4227.6157108947455,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/mean":-2.658367156982422e-05,"train/train/tensor_act_model_layers_72_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/norm":4.40625,"train/train/tensor_act_model_layers_68_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/std":3.0016993278454328e-05,"train/train/tensor_act_model_layers_2_self_attn_k_proj/max_abs":4.40625,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/std":0.031005859375,"train/train/tensor_act_model_layers_27_self_attn_v_proj/mean":0.00702667236328125,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/std":1.7034307604920187e-05,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/std":5.895625547996256e-05,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_k_proj/std":0.7832057999210864,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/mean":7.82012939453125e-05,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/max_abs":0.2255859375,"train/train/tensor_act_model_layers_63_mlp/mean":-0.003841400146484375,"train/train/tensor_act_model_layers_84_self_attn_k_proj/max_abs":5.78125,"train/train/layer__model_layers_39/param/mean":0.0013805276332147036,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/max_abs":0.1708984375,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/norm":4.375,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/max_abs":0.000499725341796875,"train/train/layer__model_layers_67/param/mean":0.001321730115298362,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/norm":0.02985062205498743,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_post_attention_layernorm/std":1.0000000745058033,"train/train/tensor_act_model_layers_89_input_layernorm/max_abs":5.3125,"train/train/tensor_act_model_layers_75_mlp/std":0.15502991563568388,"train/train/layer__model_layers_14/param/frac_near_user_limit":0,"train/train/layer_model_layers_50/act/norm":13829.011422131158,"train/train/tensor_act_model_layers_16_self_attn/mean":-0.0004601478576660156,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/max_abs":0.2353515625,"train/train/layer_model_layers_21/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/std":0.043212890625,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_input_layernorm/norm":5792.612670899491,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/norm":0.012146379474359278,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/norm":0.014476787682730944,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/max_abs":0.15625,"train/train/tensor_act_model_layers_20_self_attn_k_proj/mean":0.0936279296875,"train/train/layer__model_layers_45/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_post_attention_layernorm/mean":0.0562744140625,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/mean":4.935264587402344e-05,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/norm":0.032718942869359384,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/std":0.05078125,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/norm":5.03125,"train/train/tensor_act_model_layers_79_mlp_down_proj/mean":0.0059051513671875,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_o_proj/max_abs":0.44140625,"train/train/tensor_act_model_layers_87_mlp_up_proj/max_abs":5.25,"train/train/tensor_act_model_layers_89_mlp_up_proj/norm":7745.14421454258,"train/train/tensor_act_model_layers_22_self_attn/mean":-0.00142669677734375,"train/train/layer_model_layers_10/grad/std":4.727591367941567e-05,"train/train/tensor_act_model_layers_75_input_layernorm/norm":5792.610839850489,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/max_abs":0.000217437744140625,"train/train/tensor_act_model_layers_0_self_attn_o_proj/max_abs":0.32421875,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/std":3.5367113224513345e-05,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/norm":0.022162906556076917,"train/train/layer_model_layers_25/act/mean":-0.01682310837965745,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/max_abs":0.115234375,"train/train/tensor_act_model_layers_37_self_attn_q_proj/norm":5747.0077567076805,"train/train/layer__model_layers_78/param/std":0.057848716898753004,"train/train/layer_model_layers_50/act/max_abs":15.5,"train/train/tensor_param_model_layers_77_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/norm":3.421875,"train/train/tensor_param_model_layers_40_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/max_abs":0.0017547607421875,"train/train/tensor_act_model_layers_45_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_25/act/max_abs":17.75,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/norm":7.3125,"train/train/layer_model_layers_79/act/max_abs":11.8125,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/max_abs":0.1357421875,"train/train/tensor_act_model_layers_19_mlp_down_proj/std":0.03790302245849343,"train/train/tensor_act_model_layers_0_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/std":0.03076171875,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_mlp/max_abs":0.388671875,"train/train/tensor_act_model_layers_12_self_attn/std":0.026341273995517524,"train/train/tensor_act_model_layers_89_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78/max_abs":12.125,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn_o_proj/std":0.17285466178035025,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/max_abs":0.000659942626953125,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/max_abs":0.1298828125,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/max_abs":0.00011539459228515625,"train/train/tensor_act_model_layers_35_post_attention_layernorm/norm":5792.609497073239,"train/train/layer_model_layers_40/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/max_abs":0.1591796875,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/std":0.0458984375,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/norm":5.15625,"train/train/layer_model_layers_51/grad/max_abs":0.000919342041015625,"train/train/tensor_act_model_layers_11_mlp_up_proj/norm":2161.110628730286,"train/train/tensor_act_model_layers_13_mlp_down_proj/norm":268.95846868928186,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/std":0.0257568359375,"train/train/layer_model_layers_45/act/std":0.6723346380977128,"train/train/tensor_act_model_layers_70_mlp_down_proj/max_abs":1.453125,"train/train/tensor_param_model_layers_80_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp_down_proj/norm":1086.6291796583819,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_33/grad/norm":0.03249864467298084,"train/train/tensor_act_model_layers_18_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_k_proj/max_abs":4.96875,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/mean":-1.0058283805847168e-06,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/mean":-0.000606536865234375,"train/train/tensor_act_model_layers_67_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/max_abs":0.0006103515625,"train/train/tensor_act_model_layers_53_mlp_up_proj/max_abs":2.75,"train/train/tensor_act_model_layers_50_self_attn_v_proj/mean":0.0004131793975830078,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/mean":0.000240325927734375,"train/train/tensor_act_model_layers_53_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/max_abs":0.00013446807861328125,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/mean":-1.2278556823730469e-05,"train/train/tensor_act_model_layers_24_self_attn_o_proj/std":0.07727164999452271,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_o_proj/max_abs":1.2734375,"train/train/tensor_act_model_layers_4/norm":8668.487235294844,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/std":0.0277099609375,"train/train/tensor_act_model_layers_70_self_attn/mean":-0.002994537353515625,"train/train/tensor_act_model_layers_55_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_60/grad/max_abs":0.00128173828125,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_48/param/max_abs":1,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/grad/mean":-4.348302966161005e-07,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_up_proj/norm":2192.499040043776,"train/train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/max_abs":0.0014495849609375,"train/train/tensor_act_model_layers_52_self_attn_o_proj/mean":-0.0006465911865234375,"train/train/layer_model_layers_19/act/std":0.6448228472447343,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/mean":-9.51811671257019e-07,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/norm":0.012746889541385751,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_q_proj/mean":-0.012786865234375,"train/train/tensor_act_model_layers_73_mlp/max_abs":2.0625,"train/train/tensor_act_model_layers_85_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_88/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/std":3.140844481753664e-05,"train/train/tensor_act_model_layers_14_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/mean":-1.1981464922428131e-06,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/mean":-6.51925802230835e-09,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/max_abs":0.00014400482177734375,"train/train/tensor_act_model_layers_87_self_attn_o_proj/mean":-0.007049560546875,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/norm":5.90625,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/max_abs":0.000946044921875,"train/train/layer_model_layers_77/grad/norm":0.07216690080269351,"train/train/tensor_act_model_layers_93_self_attn_v_proj/mean":0.0067596435546875,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_q_proj/mean":0.01605224609375,"train/train/tensor_act_model_layers_52_input_layernorm/std":1.0000001545995356,"train/train/tensor_act_model_layers_28_self_attn_v_proj/std":0.30664106396701396,"train/train/layer_model_layers_87/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_mlp_up_proj/max_abs":2.609375,"train/train/layer_model_layers_37/act/mean":0.005779559795673077,"train/train/tensor_param_model_layers_10_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_88_input_layernorm/std":0.9960950888830424,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp/mean":0.0007152557373046875,"train/train/tensor_act_model_layers_3_self_attn_k_proj/std":1.4687500443864367,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp_up_proj/norm":4442.780670090519,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/mean":9.059906005859375e-05,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/max_abs":0.000766754150390625,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/mean":5.7443976402282715e-06,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_mlp_down_proj/std":0.37988416621690385,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/std":1.5535233446971765e-05,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/max_abs":0.1953125,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/std":0.031005859375,"train/train/layer_model_layers_69/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp/max_abs":6.0625,"train/train/tensor_act_model_layers_69_mlp_up_proj/mean":-0.098876953125,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/mean":-6.461050361394882e-08,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/std":3.5969880720685306e-05,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/norm":5.75,"train/train/tensor_act_model_layers_4_self_attn_q_proj/mean":0.0623779296875,"train/train/tensor_act_model_layers_66_post_attention_layernorm/max_abs":6.03125,"train/train/tensor_act_model_layers_67_mlp_up_proj/max_abs":4.03125,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp_up_proj/norm":3551.845864542114,"train/train/tensor_act_model_layers_44_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/norm":6.59375,"train/train/tensor_act_model_layers_68/max_abs":13.1875,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/norm":0.0040621707970859575,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/grad/max_abs":0.001495361328125,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/mean":1.3818498700857162e-07,"train/train/tensor_act_model_layers_66_self_attn_v_proj/mean":-0.002262115478515625,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/std":0.00019122680561083758,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/std":3.4893341914291726e-05,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_mlp_up_proj/mean":-0.103759765625,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/std":0.055908203125,"train/train/tensor_act_model_layers_36_mlp_down_proj/mean":0.00257110595703125,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/norm":0.0012338997863210407,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/norm":0.0007316167697948277,"train/train/tensor_act_model_layers_34_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/max_abs":0.0013427734375,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/max_abs":0.0004119873046875,"train/train/tensor_act_model_layers_24_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_44/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_o_proj/norm":498.73968534387893,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/max_abs":0.318359375,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_o_proj/max_abs":1.2890625,"train/train/tensor_param_model_layers_61_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_k_proj/std":0.7685619739250082,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/mean":0.00014019012451171875,"train/train/tensor_act_model_layers_71/std":1.4902410582046337,"train/train/tensor_act_model_layers_56_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90/std":2.308607898546164,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/std":4.7073605480307325e-05,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/std":0.00016509771116363775,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/norm":0.0013043252697653176,"train/train/tensor_act_model_layers_86_post_attention_layernorm/max_abs":5.28125,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_23/param/norm":19.926905002251353,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/mean":0.0003147125244140625,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_55/param/max_abs":1,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/norm":5.53125,"train/train/tensor_act_model_layers_87_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/std":7.858459254569082e-05,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/max_abs":0.0001678466796875,"train/train/tensor_act_model_layers_81_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/std":5.505001766223304e-05,"train/train/tensor_act_model_layers_30_input_layernorm/max_abs":6.09375,"train/train/tensor_act_model_layers_4_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/mean":-1.1362135410308838e-06,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/std":0.00010840377475347988,"train/train/tensor_act_model_layers_52_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/mean":1.7145066522061825e-07,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_28_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp_down_proj/max_abs":0.478515625,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/norm":0.01546446466330586,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/max_abs":0.000522613525390625,"train/train/tensor_act_model_layers_14_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92/std":2.7734511146748524,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/mean":-0.00014781951904296875,"train/train/tensor_param_model_layers_20_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_k_proj/max_abs":4.65625,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn/norm":334.2355866445592,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_67/act/mean":-0.011350154876708984,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/max_abs":0.2451171875,"train/train/tensor_act_model_layers_30_self_attn_o_proj/mean":-0.00384521484375,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_mlp_up_proj/mean":-0.135986328125,"train/train/tensor_param_model_layers_18_input_layernorm_weight/std":0,"train/train/layer__model_layers_58/param/std":0.053446129350418765,"train/train/tensor_act_model_layers_25_self_attn_v_proj/mean":0.002826690673828125,"train/train/layer_model_layers_16/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp_down_proj/max_abs":1.0390625,"train/train/tensor_act_model_layers_2_mlp_down_proj/std":0.1041261945769162,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/std":0.026123046875,"train/train/tensor_act_model_layers_84_self_attn_v_proj/norm":2526.165744285553,"train/train/tensor_param_model_layers_51_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/mean":0.000701904296875,"train/train/tensor_act_model_layers_47_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_down_proj/std":0.03643816320099371,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/max_abs":9.632110595703125e-05,"train/train/tensor_act_model_layers_34_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_12_self_attn_q_proj/max_abs":4.875,"train/train/tensor_act_model_layers_50/max_abs":15.5,"train/train/tensor_act_model_layers_60_mlp/mean":0.0005970001220703125,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/max_abs":0.00106048583984375,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/max_abs":0.0005035400390625,"train/train/layer__model_layers_14/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/std":1.3679873584548811e-05,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/norm":0.005330408499135533,"train/train/tensor_act_model_layers_56_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_34/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_up_proj/max_abs":3.328125,"train/train/tensor_act_model_layers_44_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/max_abs":0.0019378662109375,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/max_abs":0.00021266937255859375,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/norm":5.5625,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/mean":0.00017642974853515625,"train/train/tensor_param_model_layers_57_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_o_proj/mean":-0.001529693603515625,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/norm":0.0008561871101621346,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/max_abs":0.0014190673828125,"train/train/layer__model_layers_79/param/max_abs":1,"train/train/tensor_act_model_layers_66_mlp/max_abs":1.1796875,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/mean":-2.9336661100387573e-07,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/max_abs":0.0019073486328125,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/std":6.0806514452688545e-05,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91/std":2.453153865182734,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/max_abs":0.259765625,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/norm":6.59375,"train/train/tensor_act_model_layers_57_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_54/grad/mean":-1.3543892936568922e-06,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/norm":0.01530568355533992,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/mean":6.437301635742188e-05,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_input_layernorm/norm":5792.609252930556,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/std":0.02526958925322281,"train/train/tensor_act_model_layers_24_mlp_up_proj/mean":-0.0753173828125,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/std":0.05126953125,"train/train/tensor_act_model_layers_47_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn/std":0.22168658800875377,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/mean":-0.000339508056640625,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/norm":0.0018550100512453977,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/std":4.1075130125240744e-05,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/norm":0.03198381959612769,"train/train/layer_model_layers_51/grad/norm":0.04060330560133381,"train/train/tensor_act_model_layers_69_self_attn_v_proj/mean":-0.0009274482727050781,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/std":0.0001053171135014691,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_65/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/std":0.00010553489442940087,"train/train/tensor_act_model_layers_14_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_v_proj/std":0.438965799876414,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/std":3.321335612352663e-05,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/max_abs":0.1318359375,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/mean":-0.0001239776611328125,"train/train/tensor_act_model_layers_78_self_attn_v_proj/max_abs":3.265625,"train/train/tensor_act_model_layers_26_self_attn_o_proj/mean":-0.000698089599609375,"train/train/tensor_act_model_layers_68_post_attention_layernorm/max_abs":5.875,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/max_abs":0.0003814697265625,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/std":0.0277099609375,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/max_abs":0.0008392333984375,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_28/grad/norm":0.03961731007654196,"train/train/layer_model_layers_19/act/mean":-0.006125743572528546,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn/std":0.03894152385645313,"train/train/layer_model_layers_39/act/mean":0.0010559788117041956,"train/train/tensor_act_model_layers_75_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_20/grad/mean":-6.004253710371656e-07,"train/train/tensor_param_model_layers_26_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/max_abs":0.00130462646484375,"train/train/layer_model_layers_39/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp/std":0.04040545783332169,"train/train/tensor_act_model_layers_58_self_attn_q_proj/mean":-0.004062652587890625,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_post_attention_layernorm/max_abs":5.4375,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/max_abs":0.1796875,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/max_abs":0.000782012939453125,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/norm":0.0013762865245346258,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/norm":0.006824859089060465,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_up_proj/mean":-0.0782470703125,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/norm":6.46875,"train/train/tensor_act_model_layers_38_self_attn_q_proj/max_abs":6.875,"train/train/tensor_act_model_layers_51_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_52/grad/norm":0.03970904908905663,"train/train/layer__model_layers_1/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/mean":0.000213623046875,"train/train/tensor_act_model_layers_46_mlp/norm":449.96042114531235,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/max_abs":0.0002307891845703125,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/norm":6.34375,"train/train/tensor_act_model_layers_19_mlp_down_proj/max_abs":0.458984375,"train/train/tensor_act_model_layers_42_self_attn_k_proj/norm":4844.747625808526,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/std":0.045166015625,"train/train/layer__model_layers_88/param/mean":0.0012981396941424532,"train/train/tensor_act_model_layers_2_self_attn_q_proj/max_abs":5.03125,"train/train/tensor_act_model_layers_84_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_18_self_attn_k_proj/max_abs":4.71875,"train/train/tensor_act_model_layers_36_mlp_down_proj/std":0.06530795988545568,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/norm":2.875,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/max_abs":0.000492095947265625,"train/train/tensor_act_model_layers_68_mlp_up_proj/max_abs":3.453125,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/mean":0.0004177093505859375,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/mean":-0.0004825592041015625,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/mean":2.4097971618175507e-07,"train/train/tensor_param_model_layers_90_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_42_mlp/max_abs":0.7109375,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_58/grad/max_abs":0.0009918212890625,"train/train/tensor_act_model_layers_47_input_layernorm/max_abs":5.5625,"train/train/tensor_act_model_layers_60_post_attention_layernorm/std":1.000000104308123,"train/train/tensor_act_model_layers_20_mlp_down_proj/max_abs":0.578125,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/norm":0.02087884144643606,"train/train/layer_model_layers_22/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/mean":0.00013828277587890625,"train/train/tensor_act_model_layers_86_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/max_abs":0.00147247314453125,"train/train/tensor_act_model_layers_55_mlp_up_proj/norm":4225.729067297463,"train/train/layer_model_layers_19/grad/norm":0.024180994250606963,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/norm":0.0004318038898803093,"train/train/tensor_act_model_layers_23_mlp_down_proj/max_abs":0.72265625,"train/train/tensor_act_model_layers_75/std":1.5683673939197045,"train/train/tensor_act_model_layers_15_post_attention_layernorm/max_abs":6.0625,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/max_abs":0.1533203125,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/std":2.0577109628403008e-05,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/norm":9.625,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/max_abs":0.000377655029296875,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/norm":0.004592718076605472,"train/train/tensor_act_model_layers_82_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_self_attn_v_proj/norm":2094.675588011196,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/norm":0.013504157631465091,"train/train/layer_model_layers_16/act/norm":13723.645255429396,"train/train/tensor_act_model_layers_62_self_attn_o_proj/std":0.151857836758432,"train/train/tensor_act_model_layers_79_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/global/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/mean":1.2502074241638184e-05,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/std":0.00011254357207909408,"train/train/tensor_act_model_layers_3_self_attn_v_proj/max_abs":3,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_input_layernorm/mean":0.0660400390625,"train/train/tensor_act_model_layers_91_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/mean":-8.105416782200336e-08,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/max_abs":0.0002727508544921875,"train/train/tensor_act_model_layers_65_self_attn_o_proj/norm":725.704334990722,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/mean":0.000568389892578125,"train/train/tensor_act_model_layers_55_post_attention_layernorm/norm":5792.603271491274,"train/train/tensor_act_model_layers_70_self_attn_k_proj/std":0.9140752687313645,"train/train/tensor_act_model_layers_71_post_attention_layernorm/mean":0.0242919921875,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/max_abs":0.0009918212890625,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/mean":-0.0002651214599609375,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/std":0.02490234375,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/mean":0.0002307891845703125,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/max_abs":0.134765625,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/std":4.900454124197855e-05,"train/train/tensor_act_model_layers_40_mlp/std":0.06701693986414725,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/std":0.038330078125,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/mean":1.259148120880127e-06,"train/train/tensor_param_model_layers_51_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_6/param/max_abs":1,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/max_abs":2,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/std":0.054931640625,"train/train/tensor_act_model_layers_57_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/std":0.025390625,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/max_abs":0.000614166259765625,"train/train/tensor_act_model_layers_33_input_layernorm/max_abs":6.28125,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/max_abs":0.0004825592041015625,"train/train/tensor_act_model_layers_68_mlp_down_proj/max_abs":1.53125,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/std":3.238594553528646e-05,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn/std":0.05645898675300815,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_input_layernorm/std":1.0000004023312714,"train/train/layer_model_layers_61/grad/norm":0.0548392233676941,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_59_self_attn_q_proj/max_abs":5.53125,"train/train/layer_model_layers_21/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn/max_abs":1.046875,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/mean":0.0001735687255859375,"train/train/layer__model_layers_60/param/max_abs":1,"train/train/layer__model_layers_69/param/max_abs":1,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38/std":1.3046994408614985,"train/train/tensor_act_model_layers_47_self_attn/std":0.17114460530862655,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/std":0.0255126953125,"train/train/tensor_act_model_layers_54_self_attn_q_proj/mean":-0.04754638671875,"train/train/tensor_act_model_layers_42_self_attn/norm":393.05543529869465,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_33_mlp/max_abs":0.478515625,"train/train/tensor_act_model_layers_91_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp_down_proj/norm":962.1672971009383,"train/train/tensor_act_model_layers_19_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp/mean":0.00341033935546875,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/std":0.02685546875,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/std":0.00010650966553985387,"train/train/tensor_act_model_layers_60_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/max_abs":0.00017833709716796875,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/std":2.747138002397038e-05,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/mean":-4.649162292480469e-06,"train/train/tensor_act_model_layers_21_self_attn/norm":355.72957544015867,"train/train/tensor_act_model_layers_63_post_attention_layernorm/std":1.0000000949948982,"train/train/tensor_act_model_layers_37/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/norm":5.25,"train/train/tensor_act_model_layers_3_self_attn/norm":721.508086273818,"train/train/tensor_act_model_layers_43_self_attn_k_proj/norm":5032.245809096879,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/max_abs":0.2001953125,"train/train/tensor_act_model_layers_46_post_attention_layernorm/mean":0.0672607421875,"train/train/tensor_act_model_layers_67_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/std":0.031494140625,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/std":0.0242919921875,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/norm":7.34375,"train/train/layer_model_layers_37/grad/norm":0.03551613454958576,"train/train/layer_model_layers_50/grad/mean":-9.540034030882319e-07,"train/train/tensor_act_model_layers_44_self_attn_k_proj/std":0.8750000023948294,"train/train/tensor_act_model_layers_47_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/mean":7.636845111846924e-08,"train/train/tensor_act_model_layers_92_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_post_attention_layernorm/mean":0.0589599609375,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/std":0.0001059136926707406,"train/train/tensor_act_model_layers_37_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/std":8.961554533776346e-05,"train/train/tensor_param_model_layers_23_input_layernorm_weight/mean":1,"train/train/layer__model_layers_80/param/mean":0.0014930790560479842,"train/train/tensor_act_model_layers_49_post_attention_layernorm/std":1.000000104308123,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/std":0.029541015625,"train/train/tensor_act_model_layers_77_self_attn_q_proj/mean":0.05938720703125,"train/train/tensor_act_model_layers_12/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/mean":5.316734313964844e-05,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/norm":5.375,"train/train/tensor_act_model_layers_85_self_attn_v_proj/max_abs":3,"train/train/tensor_act_model_layers_18_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_23_mlp/std":0.04010023718712394,"train/train/tensor_act_model_layers_11_self_attn/max_abs":0.5546875,"train/train/layer_model_layers_78/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_k_proj/mean":0.015472412109375,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/mean":-1.0210787877440453e-06,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/mean":2.1141022443771362e-06,"train/train/tensor_act_model_layers_81_input_layernorm/max_abs":5.71875,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_q_proj/mean":0.0364990234375,"train/train/tensor_act_model_layers_29/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/norm":0.04628992522518162,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/norm":0.020599975576895524,"train/train/tensor_act_model_layers_13_self_attn/max_abs":0.462890625,"train/train/tensor_act_model_layers_29_input_layernorm/max_abs":6.25,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/norm":3.21875,"train/train/tensor_act_model_layers_21_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/mean":5.561742000281811e-08,"train/train/tensor_act_model_layers_86_mlp/max_abs":3.203125,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/max_abs":0.1337890625,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/norm":0.05768245298020056,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_29/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/norm":4.875,"train/train/tensor_act_model_layers_53_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/norm":5.6875,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/max_abs":0.00023555755615234375,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/norm":3.1875,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_51/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/mean":-6.908178329467773e-05,"train/train/tensor_act_model_layers_60_self_attn_k_proj/max_abs":6.15625,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/norm":0.007351585758062983,"train/train/tensor_act_model_layers_88_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/std":0.0291748046875,"train/train/layer_model_layers_92/grad/frac_near_user_limit":0,"train/train/layer_model_layers_27/act/max_abs":17.75,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_o_proj/norm":296.552332604813,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/std":0.04345703125,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/std":0.033203125,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_66/param/max_abs":1,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/std":5.8832803801376704e-05,"train/train/tensor_act_model_layers_58_self_attn_k_proj/max_abs":5.8125,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/max_abs":0.000141143798828125,"train/train/tensor_act_model_layers_2_self_attn_q_proj/mean":-0.002590179443359375,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/norm":0.043247704571796296,"train/train/tensor_act_model_layers_15_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_73_self_attn_k_proj/max_abs":5.25,"train/train/tensor_act_model_layers_25_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/mean":-3.0349474400281906e-07,"train/train/layer_model_layers_78/act/max_abs":12.125,"train/train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/mean":1.234002411365509e-07,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/norm":0.0018852479347227305,"train/train/tensor_act_model_layers_49_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_39/param/max_abs":1,"train/train/tensor_act_model_layers_32/std":1.3046992581231693,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/norm":0.011681537354100038,"train/train/layer__model_layers_57/param/mean":0.0014554171034028862,"train/train/tensor_act_model_layers_32_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_18/grad/std":3.9463898105753644e-05,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/max_abs":0.216796875,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/norm":0.02026174145431232,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_post_attention_layernorm/mean":0.04180908203125,"train/train/tensor_act_model_layers_52_self_attn/norm":310.7889711898425,"train/train/tensor_act_model_layers_73_self_attn/norm":307.2963042317912,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/std":8.526492292399435e-05,"train/train/tensor_act_model_layers_52_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/mean":5.537271499633789e-05,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp_down_proj/mean":0.016204833984375,"train/train/tensor_act_model_layers_29_self_attn_q_proj/max_abs":5.1875,"train/train/tensor_act_model_layers_49/mean":0.061767578125,"train/train/layer_model_layers_39/grad/std":4.4308122155988344e-05,"train/train/tensor_act_model_layers_55_mlp_down_proj/mean":-0.005828857421875,"train/train/tensor_act_model_layers_26_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/norm":7.28125,"train/train/tensor_act_model_layers_44_mlp_down_proj/std":0.06982423708988829,"train/train/tensor_act_model_layers_31_self_attn_k_proj/max_abs":4.125,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/norm":3.671875,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/norm":5.625,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/mean":4.33283275924623e-08,"train/train/tensor_act_model_layers_45_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_62/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_input_layernorm/norm":5792.609497077838,"train/train/tensor_act_model_layers_41_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/std":0.03125,"train/train/tensor_act_model_layers_72_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/std":2.6753693714756523e-05,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/max_abs":0.1689453125,"train/train/tensor_act_model_layers_81_self_attn/max_abs":2.40625,"train/train/tensor_act_model_layers_45_self_attn_k_proj/max_abs":4.59375,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_72/act/std":0.7127981588770554,"train/train/tensor_act_model_layers_33_mlp_down_proj/mean":0.0007152557373046875,"train/train/tensor_act_model_layers_49_self_attn_v_proj/std":0.3793954893340038,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_61/grad/max_abs":0.00133514404296875,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/std":2.5351472802555487e-05,"train/train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/norm":0.016702123021332452,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/norm":5.40625,"train/train/tensor_act_model_layers_1_input_layernorm/max_abs":4.46875,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/norm":0.04581362416135857,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/norm":5.625,"train/train/tensor_act_model_layers_14_mlp_down_proj/norm":267.99551695116503,"train/train/tensor_act_model_layers_44_mlp/max_abs":0.91796875,"train/train/tensor_act_model_layers_6_self_attn_v_proj/max_abs":2.5,"train/train/tensor_act_model_layers_61_mlp_down_proj/std":0.11938502695878023,"train/train/tensor_act_model_layers_18/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_down_proj/norm":319.4020248436044,"train/train/layer__model_layers_24/param/std":0.04944667002403203,"train/train/layer__model_layers_92/param/max_abs":1,"train/train/layer__model_layers_67/param/std":0.05590319172037472,"train/train/tensor_act_model_layers_47_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/norm":0.013005501940000807,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/mean":-2.066371962428093e-08,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/std":4.699793416028698e-05,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_mlp_up_proj/std":0.5312501682954409,"train/train/tensor_act_model_layers_39_mlp_down_proj/mean":-0.007293701171875,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/norm":2.9375,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/norm":5.40625,"train/train/layer_model_layers_42/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87/max_abs":12.125,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/std":0.00011076330112216436,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/max_abs":0.00115966796875,"train/train/tensor_act_model_layers_2_mlp/std":0.1041261945769162,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/max_abs":0.0021820068359375,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_2/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/std":0.03466796875,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/std":7.030827180811535e-05,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/max_abs":0.000888824462890625,"train/train/layer__model_layers_62/param/std":0.05522826494891343,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_post_attention_layernorm/max_abs":5.71875,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/mean":6.4373016357421875e-06,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/max_abs":0.12353515625,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/max_abs":0.0001583099365234375,"train/train/tensor_act_model_layers_36_self_attn_q_proj/norm":5641.622775809262,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/max_abs":0.00021076202392578125,"train/train/tensor_param_model_layers_59_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13/max_abs":18.125,"train/train/tensor_param_model_layers_33_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_36/grad/norm":0.03975023009162283,"train/train/tensor_act_model_layers_83_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/norm":8.0625,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/std":9.135558973859269e-05,"train/train/layer_model_layers_51/act/max_abs":15.375,"train/train/tensor_param_model_layers_62_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/max_abs":0.271484375,"train/train/tensor_act_model_layers_79_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/max_abs":0.000469207763671875,"train/train/tensor_param_model_layers_50_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_20_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_11_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/norm":0.024562688472500333,"train/train/tensor_act_model_layers_55_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/std":0.060791015625,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_93/grad/std":0.00012752769126399803,"train/train/tensor_act_model_layers_4_input_layernorm/mean":0.0129852294921875,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_90/param/max_abs":1,"train/train/tensor_act_model_layers_38_self_attn_v_proj/norm":2287.6514962758565,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_37/param/mean":0.0016989864164879095,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/max_abs":0.0004596710205078125,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/norm":0.004030775997845908,"train/train/layer_model_layers_41/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/mean":-3.24249267578125e-05,"train/train/tensor_act_model_layers_6_self_attn/norm":562.2322288928617,"train/train/tensor_act_model_layers_77_mlp/norm":1003.4255223609749,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/mean":-0.00019931793212890625,"train/train/layer__model_layers_40/param/max_abs":1,"train/train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/std":3.5815449840208e-05,"train/train/tensor_act_model_layers_86_self_attn_v_proj/norm":2600.294993908538,"train/train/tensor_act_model_layers_59_mlp/max_abs":1.4921875,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/mean":0.0005340576171875,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/norm":0.00032980451612816544,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/max_abs":0.000675201416015625,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/std":0.0419921875,"train/train/tensor_act_model_layers_61_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/std":4.149037427550341e-05,"train/train/layer__model_layers_34/param/max_abs":1,"train/train/layer_model_layers_26/act/norm":13708.548846986147,"train/train/tensor_act_model_layers_17_post_attention_layernorm/mean":0.0149078369140625,"train/train/tensor_act_model_layers_11_mlp_down_proj/norm":236.8761024315632,"train/train/layer__model_layers_78/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/norm":0.001590641851818361,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_o_proj/mean":-0.002048492431640625,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/std":4.191276350926896e-05,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_88_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_up_proj/frac_near_user_limit":0,"train/train/layer_model_layers_2/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/mean":-3.7670135498046875e-05,"train/train/tensor_act_model_layers_27_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_down_proj/std":0.05096452678720153,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_mlp/max_abs":0.99609375,"train/train/tensor_act_model_layers_46_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_input_layernorm/norm":5792.6059570314665,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/std":4.442941775458738e-05,"train/train/tensor_act_model_layers_47_self_attn_v_proj/max_abs":2.953125,"train/train/layer_model_layers_91/grad/max_abs":0.0015106201171875,"train/train/layer_model_layers_60/act/frac_near_user_limit":0,"train/train/layer__model_layers_86/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/norm":3.5625,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/mean":9.249895811080933e-06,"train/train/tensor_act_model_layers_35_self_attn_o_proj/std":0.12427184640003192,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/mean":-4.0372833609580994e-07,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_85/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_q_proj/mean":-0.04754638671875,"train/train/tensor_act_model_layers_58/mean":0.04742431640625,"train/train/layer__model_layers_49/param/norm":21.48414204419157,"train/train/layer__model_layers_62/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/std":2.4692260396272138e-05,"train/train/layer_model_layers_69/grad/max_abs":0.00189208984375,"train/train/tensor_act_model_layers_26_mlp_up_proj/norm":2632.647889288266,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/std":9.041795130511895e-05,"train/train/tensor_act_model_layers_2/max_abs":18.5,"train/train/layer_model_layers_74/grad/std":7.690472753328009e-05,"train/train/layer_model_layers_53/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_mlp_down_proj/norm":2199.4847827219273,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/norm":0.02602210104587484,"train/train/tensor_act_model_layers_68_self_attn_q_proj/std":0.9794941964359448,"train/train/tensor_act_model_layers_72_mlp/std":0.1635766185755264,"train/train/tensor_act_model_layers_88_mlp_up_proj/norm":7483.2042368850325,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/std":5.7118548583569255e-05,"train/train/tensor_act_model_layers_89_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/mean":-8.058547973632812e-05,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/mean":0.000362396240234375,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/norm":10.625,"train/train/layer_model_layers_59/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/max_abs":0.26953125,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/max_abs":0.00038909912109375,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/norm":6.28125,"train/train/tensor_act_model_layers_10_mlp/max_abs":0.5859375,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/norm":0.006142131716422252,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/max_abs":0.002960205078125,"train/train/tensor_act_model_layers_64_self_attn/max_abs":1.796875,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/norm":0.011381747500330924,"train/train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_rotary_emb/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/max_abs":0.0003910064697265625,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_43_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp/norm":282.2265651657156,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/norm":5.53125,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/mean":-0.00018787384033203125,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/std":6.190848001513584e-05,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/global/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/max_abs":0.00128936767578125,"train/train/layer_model_layers_36/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/std":0.025390625,"train/train/tensor_act_model_layers_40_post_attention_layernorm/mean":0.0673828125,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/std":0.031494140625,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/max_abs":0.000804901123046875,"train/train/tensor_act_model_layers_83_self_attn_o_proj/std":0.1840827039354085,"train/train/layer__model_layers_56/param/norm":21.447030042787976,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/std":0.03173828125,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/norm":5.40625,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/std":0.048828125,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/max_abs":0.000545501708984375,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/max_abs":0.00069427490234375,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/std":0.042236328125,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/norm":0.0018100085312017574,"train/train/tensor_act_model_layers_20_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/mean":1.8551945686340332e-06,"train/train/tensor_act_model_layers_23_post_attention_layernorm/norm":5792.61169433657,"train/train/tensor_act_model_layers_82_self_attn_o_proj/std":0.11097682848320566,"train/train/tensor_act_model_layers_84/frac_near_dtype_limit":0,"train/train/layer__model_layers_21/param/std":0.04942137494458747,"train/train/tensor_act_model_layers_27_mlp_up_proj/norm":2621.0368953720613,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/mean":6.314367055892944e-07,"train/train/layer_model_layers_86/grad/std":0.00010463967621794743,"train/train/tensor_act_model_layers_9_self_attn_k_proj/std":1.359375135309388,"train/train/tensor_act_model_layers_37/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp/mean":0.00499725341796875,"train/train/tensor_param_model_layers_39_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_24_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_40/max_abs":16.875,"train/train/layer_model_layers_88/act/max_abs":12.3125,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/mean":-5.5789947509765625e-05,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/norm":0.08129992785010709,"train/train/tensor_act_model_layers_22_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/std":0.028564453125,"train/train/tensor_act_model_layers_45_self_attn_o_proj/std":0.10095273818341018,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/max_abs":0.26171875,"train/train/tensor_act_model_layers_31_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/max_abs":0.19921875,"train/train/tensor_param_model_layers_19_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/norm":0.01547903186406532,"train/train/layer__model_layers_9/param/mean":0.001630756300808673,"train/train/tensor_act_model_layers_80/max_abs":11.6875,"train/train/layer_model_layers_35/act/mean":-0.0032411722036508415,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/max_abs":0.000690460205078125,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_22_self_attn_v_proj/norm":1871.5935487376155,"train/train/tensor_act_model_layers_15_input_layernorm/norm":5792.6092529312355,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/std":0.00014047602751148726,"train/train/tensor_act_model_layers_8_self_attn/max_abs":0.59375,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/max_abs":0.0004119873046875,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn_k_proj/norm":6179.668661119828,"train/train/tensor_act_model_layers_38_mlp_up_proj/norm":3122.4335380181396,"train/train/tensor_act_model_layers_29_self_attn_o_proj/max_abs":0.59375,"train/train/tensor_act_model_layers_3_input_layernorm/max_abs":4.6875,"train/train/layer_model_layers_85/grad/mean":9.716277020909299e-07,"train/train/layer_model_layers_33/grad/std":4.0114300706209095e-05,"train/train/tensor_act_model_layers_60_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/norm":0.025098143101582673,"train/train/tensor_act_model_layers_7_self_attn/max_abs":0.953125,"train/train/layer__model_layers_23/param/std":0.04918169593722733,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/norm":0.010514756145547423,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp/mean":0.012054443359375,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/max_abs":0.00014209747314453125,"train/train/tensor_act_model_layers_81_self_attn_v_proj/max_abs":2.96875,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/norm":0.03784455320102847,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/norm":4.8125,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/max_abs":0.12353515625,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/norm":10.875,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/mean":-1.4295801520347595e-06,"train/train/tensor_act_model_layers_6_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_25/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_post_attention_layernorm/std":1.0000000552972763,"train/train/tensor_act_model_layers_81_input_layernorm/mean":0.0565185546875,"train/train/tensor_act_model_layers_9_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_mlp/std":0.0827640414581374,"train/train/tensor_act_model_layers_50_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/std":3.92822617276455e-05,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/mean":-0.0001888275146484375,"train/train/tensor_param_model_layers_5_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/max_abs":0.09423828125,"train/train/layer__model_layers_30/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_87/grad/max_abs":0.001129150390625,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/max_abs":0.1953125,"train/train/tensor_param_model_layers_91_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/norm":3.34375,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/norm":0.0025773016200014622,"train/train/tensor_act_model_layers_49_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_87/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_3_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_input_layernorm_weight/mean":1,"train/train/layer__model_layers_38/param/norm":20.80886909957206,"train/train/tensor_param_model_layers_42_input_layernorm_weight/mean":1,"train/train/layer_model_layers_48/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn/max_abs":2.296875,"train/train/tensor_act_model_layers_50/norm":7494.684910749925,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/mean":-0.00038909912109375,"train/train/layer_model_layers_38/grad/norm":0.04781123048736175,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/max_abs":0.00138092041015625,"train/train/tensor_act_model_layers_44_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/std":4.3880140831032606e-05,"train/train/layer_model_layers_49/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/std":0.026123046875,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/mean":-4.630535840988159e-06,"train/train/layer_model_layers_59/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_mlp_up_proj/max_abs":2.328125,"train/train/tensor_act_model_layers_62_self_attn_v_proj/mean":-0.0006427764892578125,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/mean":6.831251084804535e-07,"train/train/layer_model_layers_65/grad/mean":-1.515251658450051e-07,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/mean":-0.0002460479736328125,"train/train/tensor_act_model_layers_29_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_33/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/norm":7.75,"train/train/tensor_act_model_layers_12_mlp_down_proj/std":0.040283203666860404,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/std":0.050537109375,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/max_abs":0.0004787445068359375,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/norm":7.1875,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/max_abs":0.0006256103515625,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/max_abs":0.0003757476806640625,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/norm":0.004000815501194026,"train/train/layer_model_layers_52/grad/max_abs":0.0008392333984375,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/max_abs":0.197265625,"train/train/layer__model_layers_80/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/std":0.034423828125,"train/train/tensor_param_model_layers_57_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/max_abs":0.001495361328125,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_k_proj/std":1.0625000560984879,"train/train/tensor_act_/mean":2.1067675352096558,"train/train/tensor_act_model_layers_65_self_attn_q_proj/norm":5873.022625142917,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/std":6.478339296296139e-05,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/std":6.915232319894934e-05,"train/train/tensor_act_model_layers_83_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp/max_abs":1.3046875,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_47/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/max_abs":0.201171875,"train/train/layer__model_layers_65/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_input_layernorm/max_abs":6.0625,"train/train/tensor_act_model_layers_46_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_45/param/norm":21.11727854447632,"train/train/tensor_act_model_layers_88_self_attn_v_proj/norm":3217.6422945638683,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_post_attention_layernorm/norm":5792.610351564465,"train/train/tensor_act_model_layers_60_self_attn_q_proj/std":0.9765669255156377,"train/train/tensor_param_model_layers_59_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/max_abs":0.000255584716796875,"train/train/tensor_act_model_layers_14_mlp_up_proj/norm":2283.816509392347,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_o_proj/mean":-0.0014362335205078125,"train/train/tensor_act_model_layers_73_input_layernorm/max_abs":5.59375,"train/train/tensor_act_model_layers_78_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_22/grad/norm":0.03818477751695803,"train/train/tensor_act_model_layers_79_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_73/mean":0.0650634765625,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/norm":5.28125,"train/train/tensor_act_model_layers_10_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_input_layernorm/norm":5792.610961915013,"train/train/tensor_act_model_layers_29_mlp_up_proj/max_abs":3.703125,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/norm":6.21875,"train/train/tensor_param_model_layers_35_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/mean":1.1315569281578064e-07,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/max_abs":0.1298828125,"train/train/tensor_act_model_layers_75_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/mean":-0.0010833740234375,"train/train/layer_model_layers_76/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/mean":-0.00021457672119140625,"train/train/tensor_act_model_layers_50_mlp_up_proj/mean":-0.09619140625,"train/train/tensor_act_model_layers_12_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_mlp_up_proj/max_abs":3.671875,"train/train/tensor_act_model_layers_82_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_k_proj/max_abs":4.78125,"train/train/tensor_act_model_layers_75_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_input_layernorm/max_abs":5.78125,"train/train/tensor_param_model_layers_49_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_79_self_attn_v_proj/std":0.4653329319328516,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/norm":0.000531579124742537,"train/train/tensor_act_model_layers_79_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_k_proj/mean":-0.01708984375,"train/train/tensor_act_model_layers_73_post_attention_layernorm/norm":5792.609375004167,"train/train/tensor_act_model_layers_68_self_attn_k_proj/norm":4846.495448970889,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/mean":-5.459785461425781e-05,"train/train/tensor_act_model_layers_26_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/norm":0.003202519577659447,"train/train/tensor_act_model_layers_31_mlp_up_proj/mean":-0.07470703125,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_16_mlp_down_proj/norm":186.95777352632697,"train/train/tensor_act_model_layers_12_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_o_proj/norm":253.85392343736257,"train/train/layer__model_layers_3/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/norm":0.06220234085198181,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/std":1.0144070058532234e-05,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/norm":0.0007574816033909324,"train/train/tensor_act_model_layers_84_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_q_proj/std":1.0781250760175152,"train/train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/max_abs":0.00084686279296875,"train/train/tensor_act_model_layers_86/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/max_abs":0.00064849853515625,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_9/act/max_abs":18.5,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/std":0.04833984375,"train/train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/max_abs":0.00054168701171875,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/norm":4.8125,"train/train/tensor_act_model_layers_37_input_layernorm/std":0.9960939519545406,"train/train/tensor_act_model_layers_78_input_layernorm/mean":0.04315185546875,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/std":0.031982421875,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/mean":0.0002899169921875,"train/train/tensor_act_model_layers_6_self_attn/max_abs":1.1171875,"train/train/tensor_act_model_layers_52_mlp/mean":0.003574371337890625,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp_down_proj/norm":885.6468437714266,"train/train/tensor_act_model_layers_90_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/norm":0.02908791087441933,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/max_abs":0.000560760498046875,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/mean":7.534027099609375e-05,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/norm":4.90625,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/norm":0.000383729203621172,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/mean":-5.453824996948242e-06,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/std":0.031494140625,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/norm":0.0006364288227643136,"train/train/tensor_param_model_layers_36_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/std":0.044189453125,"train/train/tensor_act_model_layers_64_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn_k_proj/max_abs":7.34375,"train/train/layer_model_layers_0/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/std":3.901422382329878e-05,"train/train/tensor_act_model_layers_44_self_attn_o_proj/std":0.09204155333591944,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/max_abs":0.001373291015625,"train/train/tensor_act_model_layers_19_self_attn_q_proj/norm":4839.024859254427,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_o_proj/max_abs":1.140625,"train/train/tensor_act_model_layers_40_mlp_down_proj/std":0.06701693986414725,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/std":6.259986846867153e-05,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/norm":9.8125,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/mean":-0.000396728515625,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/mean":9.387731552124023e-06,"train/train/tensor_act_model_layers_91_mlp_up_proj/norm":8344.497682473688,"train/train/layer_model_layers_31/grad/max_abs":0.00148773193359375,"train/train/tensor_act_model_layers_20_mlp/mean":0.0033111572265625,"train/train/tensor_param_model_layers_16_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_85/act/mean":-0.009839052191147437,"train/train/layer_model_layers_89/grad/max_abs":0.0014190673828125,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn_o_proj/max_abs":1.2890625,"train/train/tensor_act_model_layers_12_mlp/max_abs":0.4453125,"train/train/tensor_act_model_layers_49_self_attn_o_proj/mean":-0.00323486328125,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/norm":0.0014964605689737952,"train/train/tensor_act_model_layers_50/frac_near_dtype_limit":0,"train/train/layer_model_layers_56/act/norm":13943.968430256531,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/norm":0.017377235289395845,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/max_abs":8.153915405273438e-05,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/mean":-0.00015926361083984375,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/max_abs":0.185546875,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/mean":3.5408884286880493e-06,"train/train/tensor_act_model_layers_25_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/max_abs":0.09912109375,"train/train/tensor_act_model_layers_44/max_abs":16.5,"train/train/tensor_act_model_layers_56_self_attn_q_proj/norm":5231.118416843392,"train/train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66/norm":8324.811105742881,"train/train/tensor_act_model_layers_32_self_attn_k_proj/max_abs":4.59375,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/mean":-0.00099945068359375,"train/train/tensor_act_model_layers_65_self_attn/mean":0.0012133121490478516,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/std":0.0322265625,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/norm":0.015566178706670675,"train/train/layer__model_layers_66/param/mean":0.001376028551884263,"train/train/layer_model_layers_81/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/max_abs":0.1689453125,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/std":0.030517578125,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/max_abs":0.2216796875,"train/train/tensor_act_model_layers_76_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/max_abs":0.00079345703125,"train/train/tensor_act_model_layers_46_self_attn_k_proj/mean":-0.022705078125,"train/train/tensor_act_model_layers_51_mlp_up_proj/mean":-0.0848388671875,"train/train/tensor_act_model_layers_37_self_attn/max_abs":0.62109375,"train/train/layer__model_layers_69/param/norm":22.66719563184868,"train/train/tensor_param_model_layers_81_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_34_mlp_down_proj/norm":298.93819581491954,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/mean":1.0762596502900124e-07,"train/train/tensor_act_model_layers_86_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/norm":0.027055171238504136,"train/train/tensor_act_model_layers_60_post_attention_layernorm/mean":0.047119140625,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/norm":9.875,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/max_abs":0.0002288818359375,"train/train/tensor_act_model_layers_80_self_attn_v_proj/mean":-0.0007796287536621094,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/norm":0.006399766065608067,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/norm":4.28125,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/std":0.00011492662531735817,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_92_mlp/max_abs":10.8125,"train/train/tensor_act_model_layers_13_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/max_abs":0.000957489013671875,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/norm":0.020854650955471135,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/std":0.040283203125,"train/train/tensor_act_model_layers_25_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_83/param/mean":0.0014155442926701443,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/std":0.0001346558704337541,"train/train/tensor_act_model_layers_61_self_attn_o_proj/norm":557.7745283907466,"train/train/tensor_act_model_layers_24_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66/frac_near_dtype_limit":0,"train/train/layer_model_layers_0/act/max_abs":18,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/norm":0.029543411536040242,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/mean":-0.0001087188720703125,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/norm":4.15625,"train/train/tensor_act_model_layers_9_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_37/act/std":0.6740916654985323,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/norm":3.515625,"train/train/tensor_act_model_layers_70_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_v_proj/max_abs":1.3828125,"train/train/tensor_act_model_layers_81_mlp_down_proj/mean":0.00341033935546875,"train/train/layer__model_layers_44/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_o_proj/std":0.1251256011552929,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/max_abs":0.00016689300537109375,"train/train/layer__model_layers_51/param/mean":0.001116819575126755,"train/train/tensor_act_model_layers_77_input_layernorm/mean":0.0496826171875,"train/train/tensor_act_model_layers_67_mlp/max_abs":1.28125,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/std":4.1265489299951846e-05,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83/max_abs":11.875,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/mean":-3.498280420899391e-08,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/mean":5.9604644775390625e-05,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/max_abs":0.162109375,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52/norm":7540.220086287017,"train/train/layer_model_layers_33/grad/max_abs":0.00140380859375,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34/mean":0.081298828125,"train/train/layer__model_layers_26/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/mean":8.296966552734375e-05,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/norm":0.07470782909091943,"train/train/tensor_act_model_layers_29_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_post_attention_layernorm/max_abs":5.53125,"train/train/layer_model_layers_91/act/norm":21365.193028224716,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/mean":3.956665750592947e-08,"train/train/tensor_act_model_layers_20_self_attn/norm":130.5136547005455,"train/train/tensor_param_model_layers_71_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/mean":-7.07223080098629e-09,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/max_abs":0.0002498626708984375,"train/train/layer_model_layers_90/act/frac_near_dtype_limit":0,"train/train/layer__model_layers_36/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_62/act/std":0.7203419583364248,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/norm":0.014915729725260464,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/mean":-0.000148773193359375,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/max_abs":0.0009765625,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_q_proj/max_abs":5.21875,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/mean":4.6333298087120056e-08,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/mean":4.486646503210068e-07,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/mean":1.2114644050598145e-05,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/mean":6.693881005048752e-08,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27/frac_near_dtype_limit":0,"train/train/layer_model_layers_53/act/max_abs":15.3125,"train/train/tensor_act_model_layers_73_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/max_abs":0.1708984375,"train/train/tensor_act_model_layers_63_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/mean":1.0323856258764863e-07,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/norm":0.024573530098660652,"train/train/layer__model_layers_27/param/std":0.05009093431321509,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/mean":3.27054294757545e-09,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_92/grad/std":0.00012834847476516438,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/std":0.057373046875,"train/train/tensor_act_model_layers_27_self_attn_k_proj/max_abs":5.625,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_53/param/std":0.05293189506363443,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_43/grad/max_abs":0.00170135498046875,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/std":0.042236328125,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/mean":9.844079613685608e-07,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/mean":-5.995389074087143e-08,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/norm":0.00125943950379621,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn/mean":-0.0013980865478515625,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/std":0.0302734375,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/norm":3.546875,"train/train/tensor_act_model_layers_84_mlp/norm":1347.6208786799089,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/mean":4.601478576660156e-05,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/std":5.986728153730828e-05,"train/train/tensor_act_model_layers_75_mlp_up_proj/max_abs":3.234375,"train/train/tensor_act_model_layers_92_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/norm":6.125,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_43/grad/std":5.403130405929447e-05,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/std":0.0260009765625,"train/train/tensor_act_model_layers_61_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_91/act/mean":-0.014853550837590145,"train/train/layer_model_layers_68/act/std":0.7256261868437901,"train/train/layer__model_layers_22/param/std":0.04966812194725466,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_up_proj/mean":-0.1083984375,"train/train/tensor_act_model_layers_49_mlp/std":0.08020047800313287,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/norm":0.005960818165775771,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_post_attention_layernorm/max_abs":5.90625,"train/train/tensor_act_model_layers_92_mlp/std":0.8183711012407564,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_91/param/std":0.0626167737762589,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/std":1.1917344756844163e-05,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/std":0.0255126953125,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/mean":0.00035858154296875,"train/train/tensor_act_model_layers_79_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn_q_proj/mean":0.044189453125,"train/train/layer_model_layers_88/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/norm":7.625,"train/train/tensor_act_model_layers_85_mlp/norm":1567.4528736602604,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/std":0.03125,"train/train/tensor_act_model_layers_36_post_attention_layernorm/mean":0.0775146484375,"train/train/tensor_act_model_layers_40_self_attn_q_proj/mean":-0.16943359375,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/norm":0.0006715925941396544,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/mean":-7.534027099609375e-05,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/max_abs":6.437301635742188e-05,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/max_abs":7.200241088867188e-05,"train/train/tensor_act_model_layers_90_post_attention_layernorm/mean":0.047607421875,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/max_abs":0.2197265625,"train/train/tensor_act_model_layers_17_self_attn/norm":253.85392343736257,"train/train/tensor_act_model_layers_74_self_attn/mean":-0.0012788772583007812,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/max_abs":0.1572265625,"train/train/tensor_act_model_layers_89_self_attn_v_proj/std":0.535156424680737,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/std":0.042236328125,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_input_layernorm/norm":5792.607177737262,"train/train/layer_model_layers_86/act/mean":0.0045631115253155045,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/mean":5.587935447692871e-07,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/mean":-2.0081643015146255e-07,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/norm":5.34375,"train/train/tensor_act_model_layers_59_mlp_down_proj/norm":632.0394321420083,"train/train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_36/grad/max_abs":0.001373291015625,"train/train/tensor_act_model_layers_39_input_layernorm/norm":5792.604980474543,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/max_abs":0.000698089599609375,"train/train/tensor_act_model_layers_23_mlp/mean":0.001171112060546875,"train/train/tensor_act_model_layers_49_self_attn_q_proj/max_abs":5.4375,"train/train/tensor_act_model_layers_40/mean":0.071533203125,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_41/act/norm":13801.435413159446,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/std":5.854752559237467e-05,"train/train/tensor_act_model_layers_48_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_q_proj/max_abs":5.8125,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/max_abs":0.21484375,"train/train/tensor_act_model_layers_29_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer__model_layers_54/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/std":0.0308837890625,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/mean":-0.0009307861328125,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_14/act/mean":-0.012277733319653915,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/norm":0.012423540759410129,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/std":0.02783203125,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/norm":5.0625,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_16_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_up_proj/mean":-0.06591796875,"train/train/tensor_act_model_layers_90_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22/norm":7851.21098571715,"train/train/tensor_act_model_layers_46/std":1.3105630205640841,"train/train/tensor_act_model_layers_76_post_attention_layernorm/max_abs":5.71875,"train/train/tensor_act_model_layers_87_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_post_attention_layernorm/max_abs":5.90625,"train/train/layer__model_layers_6/param/std":0.047658259257593996,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/mean":-2.50060111284256e-07,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/mean":-4.213507054373622e-08,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_k_proj/max_abs":4.5625,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/std":0.00040607181955452926,"train/train/layer_model_layers_15/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/mean":0.00067138671875,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/mean":-2.5494955480098724e-08,"train/train/tensor_act_model_layers_90_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/norm":6.03125,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp/norm":281.0360299481865,"train/train/tensor_act_model_layers_86_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/max_abs":0.00075531005859375,"train/train/tensor_act_model_layers_93_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/mean":-9.052746463567019e-08,"train/train/tensor_act_model_layers_87_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_83/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_o_proj/std":0.26027066890145434,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/std":1.9050132263015246e-05,"train/train/tensor_param_model_layers_72_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_79_input_layernorm/max_abs":5.8125,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/norm":6.96875,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_2_self_attn_o_proj/norm":260.3362169687711,"train/train/tensor_act_model_layers_14_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_v_proj/std":0.37744258415314724,"train/train/tensor_act_model_layers_29/mean":0.0675048828125,"train/train/tensor_act_model_layers_58_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/max_abs":0.0025634765625,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/mean":1.8248101696372032e-08,"train/train/tensor_act_model_layers_55_self_attn_o_proj/norm":791.9043731140305,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/std":0.032470703125,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn/mean":-0.0006361007690429688,"train/train/tensor_act_model_layers_30/mean":0.0682373046875,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/std":0.024658203125,"train/train/layer__model_layers_5/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_q_proj/mean":-0.02569580078125,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_5/param/mean":0.0013088279878851405,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_2/grad/norm":0.04878794585343646,"train/train/tensor_act_model_layers_7/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/std":0.028076171875,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp_down_proj/max_abs":0.7890625,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/max_abs":0.00150299072265625,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn/norm":142.03019658784257,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/mean":1.0421499609947205e-06,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_down_proj/mean":0.002635955810546875,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/std":3.539814359245462e-05,"train/train/tensor_act_model_layers_57_post_attention_layernorm/norm":5792.611083986027,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/max_abs":0.1591796875,"train/train/tensor_act_model_layers_82_self_attn_q_proj/norm":5381.229349467842,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/std":9.699246119712732e-05,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/norm":0.022927212817184482,"train/train/tensor_act_model_layers_72_mlp_up_proj/norm":5089.946108893238,"train/train/tensor_act_model_layers_26_mlp/mean":0.0111846923828125,"train/train/tensor_param_model_layers_18_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/max_abs":0.0003032684326171875,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/std":5.194743262147605e-05,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/norm":4.75,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/mean":-1.8638093024492264e-07,"train/train/tensor_act_model_layers_68_self_attn_v_proj/mean":-0.0007991790771484375,"train/train/tensor_act_model_layers_27_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/norm":5.90625,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/norm":0.01843726497435695,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/max_abs":0.185546875,"train/train/tensor_act_model_layers_12/norm":8223.69542344137,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/std":0.0283203125,"train/train/tensor_act_model_layers_36_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/max_abs":0.0001239776611328125,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_26/norm":7687.166591275605,"train/train/tensor_act_model_layers_9/frac_near_dtype_limit":0,"train/train/layer__model_layers_8/param/max_abs":1,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/max_abs":0.001312255859375,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/std":2.969859689609947e-05,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/std":0.0263671875,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/norm":0.0008481398395931248,"train/train/layer_model_layers_45/grad/max_abs":0.00138092041015625,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/std":2.4459324158421958e-05,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/mean":-5.692243576049805e-06,"train/train/layer__model_layers_80/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp/mean":0.006103515625,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/max_abs":0.1298828125,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/max_abs":0.000885009765625,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_act_model_layers_50_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/std":1.2818013070890197e-05,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/mean":-8.344650268554688e-07,"train/train/layer_model_layers_80/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_k_proj/std":0.9023438408261208,"train/train/tensor_act_model_layers_88_self_attn_k_proj/max_abs":4.71875,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/norm":0.0012414832473004634,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/max_abs":0.000362396240234375,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_91/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/std":5.066701788862326e-05,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/norm":0.008932479999139762,"train/train/layer_model_layers_89/act/std":0.9467638239257895,"train/train/tensor_act_model_layers_85_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp_up_proj/std":0.5273437923855235,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/std":0.00018522561263083193,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/norm":0.07434112097768633,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/std":3.385945310738685e-05,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/max_abs":0.00069427490234375,"train/train/tensor_act_model_layers_44_self_attn_o_proj/max_abs":1.2421875,"train/train/tensor_act_model_layers_66_mlp/norm":728.174479721167,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/max_abs":0.220703125,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/std":3.7377633634020414e-05,"train/train/layer__model_layers_4/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/norm":3.421875,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/std":0.0576171875,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/std":0.036865234375,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/std":8.576557509765742e-05,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp_down_proj/norm":1070.645826143877,"train/train/layer_model_layers_51/act/std":0.6607312087718492,"train/train/tensor_act_model_layers_5/norm":8660.336379554194,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/std":0.03466796875,"train/train/tensor_act_model_layers_78_self_attn_k_proj/std":0.8476564082132358,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92/mean":0.141845703125,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/max_abs":0.0002498626708984375,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/max_abs":0.169921875,"train/train/tensor_param_model_layers_4_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_81_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp/std":0.044250955743622505,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/max_abs":0.0002002716064453125,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/norm":0.004945119058522851,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/max_abs":0.000690460205078125,"train/train/tensor_act_model_layers_49_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/mean":-2.3618340492248535e-06,"train/train/tensor_act_model_layers_15_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_up_proj/mean":-0.092529296875,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/std":0.041748046875,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/std":0.050048828125,"train/train/tensor_act_model_layers_28_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_75/param/mean":0.0010749709773547193,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/mean":0.00032806396484375,"train/train/layer__model_layers_36/param/norm":20.71507320316537,"train/train/tensor_act_model_layers_48_self_attn_o_proj/norm":406.59723534154654,"train/train/tensor_act_model_layers_24_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/norm":6.46875,"train/train/tensor_act_model_layers_2_post_attention_layernorm/max_abs":4.65625,"train/train/tensor_act_model_layers_53_input_layernorm/max_abs":5.8125,"train/train/tensor_act_model_layers_62_self_attn_k_proj/norm":5254.41903093564,"train/train/tensor_act_model_layers_24_self_attn_k_proj/std":0.9492188382115502,"train/train/tensor_act_model_layers_30_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/mean":5.838228389620781e-08,"train/train/tensor_act_model_layers_24/mean":0.0390625,"train/train/tensor_act_model_layers_46_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_89_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/max_abs":0.30078125,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19/norm":7903.410874494124,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_q_proj/norm":6145.560086243871,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/max_abs":0.00011968612670898438,"train/train/tensor_param_model_layers_52_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_47_post_attention_layernorm/max_abs":5.59375,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/max_abs":0.095703125,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/max_abs":0.248046875,"train/train/layer__model_layers_81/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_input_layernorm/max_abs":5.96875,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/norm":0.01292688731271111,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_36_self_attn/std":0.08606055346647698,"train/train/tensor_act_model_layers_46_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/norm":0.02984899964022148,"train/train/tensor_act_model_layers_17_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/norm":0.01009172519948529,"train/train/tensor_act_model_layers_32_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_input_layernorm/std":1.000000124797217,"train/train/tensor_act_model_layers_19_self_attn_o_proj/norm":142.03019658784257,"train/train/tensor_act_model_layers_20_self_attn_o_proj/mean":0.0001571178436279297,"train/train/tensor_act_model_layers_21_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/max_abs":6.914138793945312e-05,"train/train/tensor_act_model_layers_21_mlp_down_proj/max_abs":0.7109375,"train/train/tensor_act_model_layers_91_self_attn_q_proj/std":1.175787860354348,"train/train/tensor_act_model_layers_74/std":1.535163796869806,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/mean":9.94652509689331e-07,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/max_abs":0.00051116943359375,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/norm":0.0007385996710252657,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/norm":12.5625,"train/train/tensor_act_model_layers_62/mean":0.0767822265625,"train/train/tensor_act_model_layers_68_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/mean":8.89599323272705e-06,"train/train/layer_model_layers_32/grad/std":3.7618355765277404e-05,"train/train/tensor_act_model_layers_24_self_attn_q_proj/max_abs":6.3125,"train/train/tensor_act_model_layers_78_input_layernorm/max_abs":5.75,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/norm":5.4375,"train/train/layer_model_layers_17/act/mean":-0.015508798452524038,"train/train/layer_model_layers_22/act/mean":0.008509122408353366,"train/train/tensor_act_model_layers_3_self_attn_v_proj/std":0.3667002996927275,"train/train/tensor_act_model_layers_64_mlp_up_proj/norm":4750.9110155807975,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/max_abs":0.000835418701171875,"train/train/tensor_act_model_layers_21_mlp_up_proj/max_abs":3.140625,"train/train/tensor_act_model_layers_14/mean":0.0102691650390625,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_74/grad/max_abs":0.0011444091796875,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/mean":-4.38690185546875e-05,"train/train/tensor_act_model_layers_53_mlp_down_proj/std":0.08947778951549998,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/std":0.038330078125,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/max_abs":0.0009307861328125,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/norm":0.024202652165009723,"train/train/tensor_act_model_layers_42_self_attn_v_proj/mean":0.00620269775390625,"train/train/tensor_act_model_layers_12_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/norm":0.006105613348011899,"train/train/tensor_act_model_layers_45_mlp_up_proj/norm":3645.829742044641,"train/train/tensor_act_model_layers_76_mlp/std":0.16650463316070407,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/norm":0.008169425133968717,"train/train/tensor_act_model_layers_56_self_attn_o_proj/norm":326.51071662888035,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/norm":0.0008636014609335974,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_input_layernorm/mean":0.0615234375,"train/train/tensor_act_model_layers_80_mlp_up_proj/std":0.5507812905818843,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/max_abs":0.00022983551025390625,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/mean":-2.1175947040319443e-07,"train/train/tensor_act_model_layers_65_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/norm":4.96875,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/norm":3,"train/train/tensor_act_model_layers_41_mlp/std":0.06653403423864701,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/norm":3.859375,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/norm":0.0848433459080861,"train/train/tensor_act_model_layers_61_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_91/param/max_abs":1,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/max_abs":0.00127410888671875,"train/train/tensor_act_model_layers_40_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn_v_proj/std":0.2331547578275328,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/mean":8.262693881988525e-06,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_16/param/max_abs":1,"train/train/tensor_act_model_layers_61_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/mean":-1.171603798866272e-06,"train/train/tensor_act_model_layers_79/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp_down_proj/mean":0.0033111572265625,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/max_abs":0.000499725341796875,"train/train/tensor_act_model_layers_42/mean":0.07373046875,"train/train/tensor_act_model_layers_41_input_layernorm/max_abs":6,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp_down_proj/norm":583.6085677580277,"train/train/tensor_act_model_layers_16_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_7/param/std":0.0492062108068897,"train/train/tensor_act_model_layers_0_mlp/max_abs":17.875,"train/train/tensor_act_model_layers_76_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/max_abs":0.1640625,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/norm":5.5,"train/train/tensor_act_model_layers_89_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/std":8.738386479813547e-05,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/std":2.0737012401230418e-05,"train/train/tensor_act_model_layers_40_self_attn_v_proj/max_abs":2.53125,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/norm":5.71875,"train/train/tensor_act_model_layers_75_self_attn_q_proj/mean":-0.064208984375,"train/train/tensor_act_model_layers_12_input_layernorm/norm":5792.6042480520655,"train/train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn/std":0.10681408754319424,"train/train/layer_model_layers_77/act/std":0.7587531027425818,"train/train/tensor_param_model_layers_58_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/std":0.051025390625,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/max_abs":0.00017452239990234375,"train/train/tensor_act_model_layers_43_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_up_proj/norm":2860.6913139761937,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/std":0.033447265625,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/std":0.00010443240548547454,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/mean":0.00035858154296875,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/std":7.717452547224389e-05,"train/train/tensor_act_model_layers_73_post_attention_layernorm/std":1.0000020563581307,"train/train/layer_model_layers_81/act/max_abs":11.9375,"train/train/tensor_act_model_layers_69_self_attn_k_proj/norm":5103.835789327268,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/max_abs":0.0002899169921875,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/mean":-1.4513731002807617e-05,"train/train/layer_model_layers_89/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/mean":0.00011682510375976562,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_q_proj/norm":5750.004333663748,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/mean":9.250640869140625e-05,"train/train/tensor_act_model_layers_85_self_attn/std":0.13940500021524083,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/norm":0.006819859397886488,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/mean":0.00019550323486328125,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/mean":0.000255584716796875,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/max_abs":0.0001697540283203125,"train/train/tensor_act_model_layers_5_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/norm":0.05341284132762917,"train/train/tensor_act_model_layers_51_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36/std":1.3046993323606184,"train/train/tensor_act_model_layers_47/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_87_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_k_proj/norm":5019.37340814188,"train/train/tensor_act_model_layers_36_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/mean":-0.00038909912109375,"train/train/tensor_param_model_layers_66_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/std":0.0247802734375,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_v_proj/max_abs":3.828125,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_down_proj/max_abs":6.0625,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/max_abs":0.00017833709716796875,"train/train/tensor_act_model_layers_93_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_embed_tokens_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_14_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn/max_abs":1.375,"train/train/tensor_act_model_layers_52_input_layernorm/mean":0.04241943359375,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/norm":0.02926933265495459,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/mean":6.008148193359375e-05,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/max_abs":0.00026702880859375,"train/train/layer__model_layers_51/param/std":0.05326388240338251,"train/train/tensor_act_model_layers_86_mlp/std":0.24609377412568836,"train/train/layer__model_layers_14/param/std":0.04860907138658898,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/max_abs":0.00118255615234375,"train/train/tensor_act_model_layers_23_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_v_proj/max_abs":3.03125,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/mean":-1.0418079909868538e-07,"train/train/tensor_act_model_layers_61_mlp_down_proj/max_abs":1.6796875,"train/train/layer_model_layers_66/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn/norm":787.1452617548202,"train/train/layer__model_layers_10/param/std":0.04872301080720276,"train/train/tensor_act_model_layers_93_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_input_layernorm/mean":0.0697021484375,"train/train/tensor_act_model_layers_87_self_attn_k_proj/mean":0.02081298828125,"train/train/tensor_act_model_layers_45_mlp_down_proj/norm":462.3028784268603,"train/train/tensor_act_model_layers_85_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_o_proj/std":0.044922308346353024,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/norm":0.0033944804486291335,"train/train/layer_model_layers_77/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/max_abs":0.00037384033203125,"train/train/layer_model_layers_88/grad/std":0.00013220854495455645,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn/std":0.21241145607705472,"train/train/tensor_act_model_layers_28_self_attn_q_proj/max_abs":5.1875,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_22/grad/mean":-5.001502798785937e-07,"train/train/tensor_act_model_layers_82/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_80/act/std":0.8308556977363757,"train/train/tensor_act_model_layers_93_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/max_abs":0.000698089599609375,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/mean":2.283195499330759e-08,"train/train/tensor_act_model_layers_52_self_attn_k_proj/mean":-0.02386474609375,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/std":1.9380845669467115e-05,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_24/act/std":0.6924421953730404,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/norm":0.0032079222564668796,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/max_abs":0.00017547607421875,"train/train/tensor_act_model_layers_9_mlp_up_proj/max_abs":2.59375,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/norm":10.625,"train/train/tensor_act_model_layers_42_self_attn_q_proj/std":0.9863301003313586,"train/train/tensor_param_model_layers_52_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_post_attention_layernorm/max_abs":6.0625,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/std":0.041259765625,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/max_abs":0.16796875,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_4/param/mean":0.0014624156743613494,"train/train/tensor_act_model_layers_36/norm":7581.9100616824935,"train/train/layer_model_layers_74/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_self_attn_k_proj/std":0.8203172229449232,"train/train/layer_model_layers_82/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/norm":5.8125,"train/train/tensor_act_model_layers_39_post_attention_layernorm/mean":0.075927734375,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_k_proj/std":0.9267597640633009,"train/train/tensor_act_model_layers_90/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/mean":-0.00017452239990234375,"train/train/tensor_act_model_layers_42_mlp_down_proj/std":0.06250000217551129,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/mean":4.0105078369379044e-08,"train/train/tensor_act_model_layers_12_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn_v_proj/norm":1654.0310310342852,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn/mean":-0.00211334228515625,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_mlp/norm":268.95846868928186,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/mean":-1.7061829566955566e-06,"train/train/tensor_act_model_layers_58_self_attn/max_abs":1.328125,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_68/act/mean":-0.009310062115009014,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_64/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_71/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_input_layernorm/std":0.9960938322777808,"train/train/layer_model_layers_9/act/std":0.8629669915627872,"train/train/tensor_act_model_layers_80_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_k_proj/mean":0.026275634765625,"train/train/tensor_act_model_layers_59_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/mean":-2.4009495973587036e-06,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/max_abs":0.091796875,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/std":0.05322265625,"train/train/tensor_act_model_layers_45_self_attn/std":0.10095273818341018,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_5/param/max_abs":1,"train/train/tensor_act_model_layers_66_self_attn_k_proj/norm":5129.500760986864,"train/train/tensor_act_model_layers_21_mlp_up_proj/mean":-0.0621337890625,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/std":1.3731869792474196e-05,"train/train/tensor_act_model_layers_4_self_attn_o_proj/norm":285.5546347633865,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/max_abs":0.00070953369140625,"train/train/tensor_act_model_layers_70_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/std":1.668331103740251e-05,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/max_abs":0.00069427490234375,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/norm":0.0006997526402526175,"train/train/tensor_act_model_layers_38_mlp_up_proj/max_abs":2.484375,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn_k_proj/std":0.8095721807331759,"train/train/layer_model_layers_78/grad/std":9.046363433724422e-05,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/mean":-0.0002574920654296875,"train/train/tensor_act_model_layers_93_self_attn_k_proj/norm":4951.91607629873,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/std":8.12604526845366e-05,"train/train/layer_model_layers_69/grad/std":7.958256332626061e-05,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/std":0.036865234375,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/mean":-4.458427429199219e-05,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/max_abs":0.0002841949462890625,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/norm":0.009193239962124641,"train/train/tensor_act_model_layers_28_self_attn_k_proj/mean":-0.01702880859375,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/max_abs":0.0002498626708984375,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn_k_proj/std":0.8808615741575408,"train/train/tensor_param_model_layers_50_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_24_self_attn_o_proj/max_abs":0.71484375,"train/train/tensor_act_model_layers_84_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_16/grad/max_abs":0.0014495849609375,"train/train/tensor_act_model_layers_76_self_attn_v_proj/std":0.4228528743292619,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/max_abs":0.000347137451171875,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/norm":0.012735011313051098,"train/train/tensor_act_model_layers_33_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_78/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/std":0.029541015625,"train/train/tensor_act_model_layers_30_self_attn_k_proj/mean":-0.01324462890625,"train/train/tensor_act_model_layers_20_mlp/max_abs":0.578125,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/max_abs":0.0015106201171875,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/mean":0.00015926361083984375,"train/train/layer__model_layers_6/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_mlp/max_abs":0.5390625,"train/train/layer__model_layers_64/param/norm":22.87982360345027,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/std":0.0284423828125,"train/train/tensor_act_model_layers_0_mlp_down_proj/mean":0.0394287109375,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn/max_abs":0.57421875,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/norm":3.828125,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/mean":-0.00063323974609375,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/mean":-1.768785296007991e-08,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/max_abs":0.000469207763671875,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/mean":1.6050762496888638e-08,"train/train/tensor_act_model_layers_57_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/norm":0.015007482539952694,"train/train/tensor_act_model_layers_62_post_attention_layernorm/mean":0.0577392578125,"train/train/tensor_act_model_layers_80_self_attn_o_proj/std":0.23828528167658544,"train/train/layer_model_layers_26/act/std":0.6564395568811673,"train/train/tensor_act_model_layers_42_mlp_down_proj/mean":0.0007123947143554688,"train/train/layer_model_layers_4/grad/max_abs":0.00148773193359375,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/norm":2042.8077932438202,"train/train/tensor_act_model_layers_92/norm":16084.591245892496,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/mean":-9.632110595703125e-05,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/mean":7.846392691135406e-07,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/norm":4.1875,"train/train/tensor_act_model_layers_70/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/max_abs":0.134765625,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/norm":5.65625,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/norm":0.008136898599176242,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/std":0.03466796875,"train/train/layer__model_layers_28/param/max_abs":1,"train/train/tensor_act_model_layers_50_self_attn_k_proj/norm":4526.1068366943855,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/norm":3.703125,"train/train/tensor_act_model_layers_50_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_down_proj/max_abs":0.427734375,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/std":4.9055482237774104e-05,"train/train/tensor_act_model_layers_8_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp_down_proj/std":0.07739296574224763,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/max_abs":0.12109375,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/mean":0.00014495849609375,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/std":0.02294921875,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/std":0.03369140625,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/norm":3.375,"train/train/layer__model_layers_88/param/max_abs":1,"train/train/layer__model_layers_33/param/max_abs":1,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/mean":1,"train/train/layer__model_layers_44/param/norm":21.268925395985573,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_self_attn_o_proj/norm":643.2306815338605,"train/train/tensor_act_model_layers_60_input_layernorm/mean":0.05010986328125,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/max_abs":0.000244140625,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/mean":-2.600252628326416e-06,"train/train/tensor_param_model_layers_26_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_k_proj/norm":5877.957157978943,"train/train/tensor_act_model_layers_21_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_post_attention_layernorm/norm":5792.607177739278,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/norm":0.03830538397662899,"train/train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp_up_proj/max_abs":3.71875,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_input_layernorm/max_abs":6.21875,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/max_abs":0.224609375,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/max_abs":0.0003032684326171875,"train/train/tensor_act_model_layers_87_mlp/mean":0.016998291015625,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/max_abs":0.0003223419189453125,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/norm":0.003169873414694554,"train_runtime":3719.287,"train/train/tensor_act_model_layers_4_mlp_down_proj/max_abs":0.53125,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/max_abs":0.158203125,"train/train/tensor_act_model_layers_91_self_attn_v_proj/norm":2967.6289493243426,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/norm":0.005858460990824887,"train/train/tensor_act_model/max_abs":5.25,"train/train/tensor_act_model_layers_55_self_attn_q_proj/norm":5931.890202964324,"train/train/tensor_act_model_layers_80_mlp_up_proj/mean":-0.1129150390625,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29/max_abs":17.875,"train/train/tensor_act_model_layers_65_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/std":0.0289306640625,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_25/grad/std":4.457498735993078e-05,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn_k_proj/max_abs":4.75,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/max_abs":0.25390625,"train/train/layer_model_layers_90/grad/max_abs":0.0012359619140625,"train/train/tensor_act_model_layers_56_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_input_layernorm/norm":5792.606811525222,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/norm":0.02468508020444763,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn/mean":0.0012607574462890625,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/mean":5.476176738739014e-07,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/std":7.246636651105891e-05,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/mean":-5.650520324707031e-05,"train/train/layer_model_layers_30/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_post_attention_layernorm/std":0.996096165972005,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/mean":-0.0003910064697265625,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp_up_proj/norm":2760.8555533654235,"train/train/tensor_act_model_layers_79_self_attn_v_proj/max_abs":3.828125,"train/train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_input_layernorm/mean":0.04791259765625,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/mean":-9.830109775066376e-07,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/norm":0.0221023125034125,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn_v_proj/mean":0.00618743896484375,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/std":1.388647792331504e-05,"train/train/layer__model_layers_22/param/norm":20.123337955437464,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/mean":2.816086634993553e-07,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/norm":0.0015355803122988367,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_input_layernorm/std":1.0000006612388095,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/mean":0.000545501708984375,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/mean":-8.975621312856674e-07,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/norm":5.71875,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/std":6.319449441946016e-05,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/norm":0.0020348075800949808,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/std":3.891453397563353e-05,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/mean":-9.080395102500916e-08,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/std":6.402139342746327e-05,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/norm":0.008177629086126641,"train/train/tensor_act_model_layers_38_mlp_down_proj/max_abs":0.67578125,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_67/grad/max_abs":0.00113677978515625,"train/train/tensor_act_model_layers_86_self_attn_v_proj/std":0.4487314097111314,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/norm":0.0007123838625570345,"train/train/tensor_param_model_layers_36_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_o_proj/max_abs":0.8125,"train/train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_51/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn/max_abs":0.8125,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/std":0.0238037109375,"train/train/tensor_act_model_layers_34/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/mean":0.00010991096496582031,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/mean":-5.888938903808594e-05,"train/train/tensor_act_model_layers_62_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_60_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_30/param/std":0.05017436229202368,"train/train/tensor_act_model_layers_72_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25/std":1.3340001307795546,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/norm":6.59375,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/norm":0.0024749125556116475,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_up_proj/std":0.7675807263667553,"train/train/tensor_act_model_layers_48_self_attn_q_proj/norm":5325.00054191454,"train/train/tensor_act_model_layers_82_mlp/norm":1306.9650487676554,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/max_abs":0.1435546875,"train/train/tensor_act_model_layers_42_post_attention_layernorm/max_abs":6.03125,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/max_abs":0.24609375,"train/train/layer_model_layers_47/act/std":0.6999196518025654,"train/train/tensor_act_model_layers_32_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/grad/mean":-5.70730019982631e-07,"train/train/tensor_act_model_layers_52_mlp_up_proj/norm":4021.477601815953,"train/train/tensor_act_model_layers_38_mlp/max_abs":0.67578125,"train/train/tensor_act_model_layers_66_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_input_layernorm/std":1.0000000100990292,"train/train/tensor_act_model_layers_9_self_attn_v_proj/mean":0.0153350830078125,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/mean":-1.8317223293706775e-07,"train/train/tensor_act_model_layers_70_post_attention_layernorm/std":1.0000018938426622,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/max_abs":0.000934600830078125,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/max_abs":0.2314453125,"train/train/tensor_act_model_layers_71/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/max_abs":0.15234375,"train/train/tensor_act_model_layers_67_self_attn/norm":551.3289695599515,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/norm":7.1875,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/max_abs":0.11181640625,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_mlp_up_proj/max_abs":4,"train/train/tensor_act_model_layers_78_post_attention_layernorm/norm":5792.608398437798,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_75/act/mean":-0.014247435789841872,"train/train/tensor_act_model_layers_33_post_attention_layernorm/std":0.9960938322777808,"train/train/tensor_act_model_layers_40_mlp/max_abs":0.9296875,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/std":0.04931640625,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/max_abs":0.000213623046875,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_q_proj/mean":-0.178466796875,"train/train/tensor_act_model_layers_87_post_attention_layernorm/max_abs":5.5,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/std":0.0269775390625,"train/train/tensor_act_model_layers_6_self_attn/mean":-0.0014362335205078125,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/mean":-1.7334241420030594e-07,"train/train/tensor_act_model_layers_22_self_attn_q_proj/max_abs":5.15625,"train/train/tensor_param_model_layers_24_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_55_mlp_down_proj/max_abs":0.99609375,"train/train/tensor_act_model_layers_11/max_abs":18.25,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/max_abs":0.00013637542724609375,"train/train/tensor_act_model_layers_62/std":1.371099504638072,"train/train/tensor_act_model_layers_14_self_attn_o_proj/mean":-0.00021070241928100586,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/std":0.036376953125,"train/train/tensor_act_model_layers_77_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/mean":7.927417755126953e-06,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/norm":6.3125,"train/train/tensor_param_model_layers_42_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/norm":11.3125,"train/train/global/grad/mean":-1.663670871785044e-07,"train/train/tensor_act_/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/mean":-6.467336788773537e-06,"train/train/tensor_act_model_layers_48_self_attn_o_proj/max_abs":1.6484375,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_input_layernorm/norm":5792.611572265857,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn_v_proj/max_abs":1.96875,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/std":9.314865307772293e-05,"train/train/layer_model_layers_93/grad/max_abs":0.00147247314453125,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/mean":-3.329478204250336e-08,"train/train/tensor_act_model_layers_22_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_q_proj/norm":5726.715683638844,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/std":1.0688422345281125e-05,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/std":5.664768382021403e-05,"train/train/layer__model_layers_80/param/norm":23.76522949210674,"train/train/layer__model_layers_19/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_v_proj/max_abs":2.5625,"train/train/tensor_act_model_layers_24_mlp_down_proj/std":0.04852337578370075,"train/train/tensor_act_model_layers_83_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/std":0.0255126953125,"train/train/tensor_act_model_layers_55_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/mean":0.00017261505126953125,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/mean":1.0128132998943329e-06,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/mean":3.4570693969726562e-06,"train/train/tensor_act_model_layers_89_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_o_proj/max_abs":1.2890625,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/mean":5.139736458659172e-07,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/max_abs":0.1337890625,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/max_abs":0.00012111663818359375,"train/train/tensor_act_model_layers_79_mlp_down_proj/norm":1145.4421652145625,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/mean":0.000217437744140625,"train/train/tensor_act_model_layers_61_post_attention_layernorm/norm":5792.607910161981,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp_down_proj/max_abs":1.125,"train/train/tensor_act_model_layers_63_self_attn_q_proj/std":1.0156269807062643,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/max_abs":0.23046875,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/norm":0.00669196631564607,"train/train/tensor_act_model_layers_77_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/mean":-0.000278472900390625,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/norm":5.875,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_62/param/norm":22.378665900808297,"train/train/tensor_act_model_layers_13/norm":8161.451156938056,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/mean":-0.00013065338134765625,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/mean":-1.4901161193847656e-05,"train/train/tensor_act_model_layers_33_mlp/norm":297.5330767104795,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/max_abs":0.1591796875,"train/train/tensor_act_model_layers_84_self_attn_k_proj/mean":-0.0093536376953125,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/mean":0.0003032684326171875,"train/train/tensor_act_model_layers_86_self_attn_k_proj/max_abs":4.625,"train/train/tensor_act_model_layers_68_self_attn_o_proj/norm":1043.7661422415608,"train/train/tensor_act_model_layers_31_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_27/grad/max_abs":0.001739501953125,"train/train/layer_model_layers_11/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn_o_proj/std":0.156739104606625,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_43/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/mean":-0.0002994537353515625,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/std":0.0400390625,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/mean":-5.587935447692871e-09,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/std":0.0001303050885244046,"train/train/layer__model_layers_80/param/max_abs":1,"train/train/layer_model_layers_64/act/std":0.7143072658137206,"train/train/tensor_act_model_layers_72_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_self_attn_q_proj/mean":0.04901123046875,"train/train/tensor_act_model_layers_11_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/std":0.00014024250680491016,"train/train/tensor_act_model_layers_3_self_attn_k_proj/max_abs":6.90625,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/norm":0.0015749711177487167,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/norm":6.1875,"train/train/tensor_param_model_layers_69_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_39_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/max_abs":0.1279296875,"train/train/tensor_act_model_layers_71_self_attn_o_proj/std":0.07840758077749085,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/std":2.6665793337032023e-05,"train/train/tensor_act_model_layers_34_self_attn_q_proj/mean":-0.017669677734375,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/norm":0.019970158427302015,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/norm":0.015614774457445963,"train/train/tensor_act_model_layers_36_self_attn/norm":498.73968534387893,"train/train/tensor_act_model_layers_33_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/mean":1.3585668057203293e-06,"train/train/tensor_act_model_layers_62_mlp/std":0.11377005852924488,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/mean":0.00018024444580078125,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_91_self_attn_o_proj/mean":-0.0016422271728515625,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/max_abs":0.00078582763671875,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/norm":0.016175813425709628,"train/train/tensor_act_model_layers_81_mlp_down_proj/max_abs":2.234375,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/mean":-0.001556396484375,"train/train/layer__model_layers_89/param/std":0.06232374136348162,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/max_abs":0.1845703125,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/norm":6.8125,"train/train/tensor_act_model_layers_11_self_attn_o_proj/max_abs":0.5546875,"train/train/tensor_act_model_layers_15_input_layernorm/max_abs":6.0625,"train/train/tensor_param_model_layers_55_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/norm":0.002117047006311006,"train/train/tensor_act_model_layers_41_self_attn_v_proj/std":0.30859379269936876,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_75_self_attn_k_proj/std":0.912111504938224,"train/train/layer_model_layers_33/act/norm":13790.672453342682,"train/train/tensor_act_model_layers_6_input_layernorm/max_abs":5.28125,"train/train/tensor_act_model_layers_40_post_attention_layernorm/std":0.9960940491918975,"train/train/tensor_act_model_layers_80_self_attn_o_proj/max_abs":2.625,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/std":9.172318322418187e-05,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/norm":0.014988014976397011,"train/train/layer_model_layers_83/act/std":0.8287122879569792,"train/train/tensor_act_model_layers_74_self_attn/max_abs":1.1171875,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/max_abs":0.1220703125,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_35/act/max_abs":17.625,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/norm":5.375,"train/train/tensor_act_model_layers_91_post_attention_layernorm/max_abs":5.1875,"train/train/tensor_act_model_layers_69_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/mean":-3.75211238861084e-05,"train/train/tensor_param_model_layers_17_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_29_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_53/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/std":2.870624830291672e-05,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/max_abs":0.11181640625,"train/train/tensor_act_model_layers_9_input_layernorm/mean":-0.00696563720703125,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_mlp_down_proj/norm":378.6767679017914,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14/norm":8120.696981451698,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/norm":0.014622283475403982,"train/train/tensor_act_model_layers_84/frac_near_user_limit":0,"train/train/tensor_act_model_embed_tokens/norm":642.613959402126,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/mean":4.5262277126312256e-07,"train/train/tensor_act_model_layers_34_self_attn_o_proj/max_abs":1.2109375,"train/train/layer__model_layers_59/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_10/param/mean":0.0017354789650570399,"train/train/tensor_act_model_layers_53_post_attention_layernorm/norm":5792.610717774039,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/std":0.0284423828125,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/max_abs":0.1611328125,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_47/grad/max_abs":0.002105712890625,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_post_attention_layernorm/std":1.0000000386498862,"train/train/tensor_act_model_layers_49_mlp_up_proj/norm":3834.292344798665,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn_q_proj/std":0.9902365569743358,"train/train/tensor_act_model_layers_68_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/norm":0.009683165503446544,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/max_abs":0.0005035400390625,"train/train/tensor_act_model_layers_52_post_attention_layernorm/max_abs":5.90625,"train/train/tensor_act_model_layers_93_post_attention_layernorm/std":1.0000007357445635,"train/train/tensor_act_model_layers_76_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/std":0.02294921875,"train/train/tensor_act_model_layers_73_mlp_up_proj/norm":5070.005659032578,"train/train/tensor_act_model_layers_66_self_attn_k_proj/max_abs":5.1875,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/norm":0.000502780862195027,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/norm":0.015143423748870918,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/std":0.0245361328125,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/mean":-0.00020503997802734375,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/mean":-0.000125885009765625,"train/train/tensor_act_model_layers_5_self_attn_k_proj/max_abs":5.3125,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/norm":5.4375,"train/train/tensor_act_model_layers_73_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_post_attention_layernorm/std":1.0000000819563832,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/std":5.01563308389198e-05,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/max_abs":0.0016021728515625,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_self_attn_k_proj/mean":0.048828125,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/norm":0.00866795923989777,"train/train/layer__model_layers_25/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_59/act/max_abs":14.1875,"train/train/tensor_act_model_layers_6_input_layernorm/norm":5792.6064453131485,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_input_layernorm/max_abs":5.75,"train/train/tensor_act_model_layers_16_self_attn_q_proj/max_abs":4.90625,"train/train/tensor_act_model_layers_64_post_attention_layernorm/max_abs":5.84375,"train/train/tensor_param_model_layers_79_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_16_self_attn_v_proj/std":0.25390626311552894,"train/train/tensor_act_model_layers_15_mlp_down_proj/max_abs":0.59375,"train/train/tensor_act_model_layers_66_post_attention_layernorm/std":1.0000007525083572,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/std":0.030029296875,"train/train/tensor_act_model_layers_71_post_attention_layernorm/max_abs":5.875,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/norm":0.026653641124750393,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/std":2.258773448526473e-05,"train/train/tensor_act_model_layers_92_mlp_up_proj/norm":8725.35245872818,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/mean":-1.0700896382331848e-06,"train/train/tensor_act_model_layers_65_self_attn_k_proj/max_abs":4.96875,"train/train/tensor_act_model_layers_88_mlp/mean":-0.00015461444854736328,"train/train/tensor_act_model_layers_59_post_attention_layernorm/norm":5792.609375001172,"train/train/tensor_act_model_layers_76_mlp/norm":962.1672971009383,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/max_abs":0.00055694580078125,"train/train/tensor_act_model_layers_66_mlp_up_proj/mean":-0.107666015625,"train/train/tensor_act_model_layers_28_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_44/grad/std":5.170144728850115e-05,"train/train/tensor_act_model_layers_47_self_attn_q_proj/std":1.0156250085968237,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_6/param/norm":19.316096196323546,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp_up_proj/std":0.21191413281698862,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/norm":4.375,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/mean":-0.0003719329833984375,"train/train/tensor_act_model_layers_52_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/max_abs":0.0001850128173828125,"train/train/layer_model_layers_8/grad/mean":-3.1018744450928453e-07,"train/train/tensor_act_model_layers_48_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/norm":0.021997734169886877,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_26/grad/norm":0.029829637537159025,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/mean":8.329516276717186e-08,"train/train/layer_model_layers_93/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/max_abs":0.0002460479736328125,"train/train/layer_model_layers_25/grad/mean":-4.947938129608047e-07,"train/train/tensor_act_model_layers_61_self_attn_v_proj/mean":0.0014514923095703125,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/std":3.763491850591301e-05,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_50/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/norm":4.28125,"train/train/tensor_act_model_layers_80_self_attn/max_abs":2.625,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/max_abs":0.2578125,"train/train/tensor_param_model_layers_44_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_36_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_46/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp/norm":696.4606801217087,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/mean":-5.085021257400513e-07,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/mean":3.7178397178649902e-06,"train/train/tensor_act_model_layers_47_mlp_up_proj/max_abs":3.296875,"train/train/tensor_act_model_layers_80_self_attn_o_proj/norm":1381.2628996475278,"train/train/tensor_act_model_layers_41_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/max_abs":0.2314453125,"train/train/tensor_act_model_layers_34_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_o_proj/max_abs":0.88671875,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn_k_proj/max_abs":4.5625,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/norm":5.59375,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/mean":-9.278301149606705e-08,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/norm":3.921875,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/norm":0.02193838201417975,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/max_abs":0.255859375,"train/train/tensor_act_model_layers_11_post_attention_layernorm/std":1.0000000217405611,"train/train/tensor_act_model_layers_47_mlp/mean":-0.00173187255859375,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/norm":4.125,"train/train/tensor_act_model_layers_37_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_47_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/max_abs":0.240234375,"train/train/tensor_act_model_layers_48_self_attn_q_proj/max_abs":5.6875,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/max_abs":0.00139617919921875,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_down_proj/max_abs":1.3984375,"train/train/tensor_param_model_layers_64_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/mean":2.3926841095089912e-06,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/mean":0.00010347366333007812,"train/train/tensor_act_model_layers_2_post_attention_layernorm/std":1.0000000093132257,"train/train/tensor_param_model_layers_36_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/mean":1,"train/train/layer__model_layers_81/param/max_abs":1,"train/train/tensor_act_model_layers_7_mlp_up_proj/mean":-0.05938720703125,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/max_abs":0.00170135498046875,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/max_abs":0.00153350830078125,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/std":0.027099609375,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/std":8.63388432066099e-05,"train/train/tensor_act_model_layers_39_self_attn/mean":0.0006642341613769531,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_25_self_attn_q_proj/max_abs":6.40625,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/max_abs":0.00072479248046875,"train/train/tensor_act_model_layers_75_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/norm":5.8125,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/max_abs":0.000461578369140625,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/norm":0.011833621324081566,"train/train/tensor_act_model_layers_20_self_attn_v_proj/norm":1411.6401213035933,"train/train/tensor_act_model_layers_25_mlp_up_proj/mean":-0.0682373046875,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp_up_proj/mean":-0.0765380859375,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/mean":0.000713348388671875,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/std":0.056884765625,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/max_abs":0.00106048583984375,"train/train/layer_model_layers_54/grad/norm":0.041843117439788156,"train/train/tensor_param_model_layers_68_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/std":0.029052734375,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn/norm":152.75751545221004,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/max_abs":0.0008392333984375,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_input_layernorm/std":0.9960939519545406,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/max_abs":0.000530242919921875,"train/train/tensor_act_model_layers_51_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/std":0.0419921875,"train/train/layer_model_layers_64/grad/std":7.77898072111298e-05,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_o_proj/std":0.07080417966085727,"train/train/tensor_act_model_layers_37_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/std":3.1931959212972455e-05,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/max_abs":0.169921875,"train/train/tensor_act_model_layers_19_mlp/norm":222.34625983155243,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/mean":0.00031280517578125,"train/train/tensor_act_model_layers_21_input_layernorm/mean":0.03173828125,"train/train/tensor_act_model_layers_17_self_attn_v_proj/mean":0.003543853759765625,"train/train/tensor_act_model_layers_55_self_attn_q_proj/std":1.023437916322434,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/std":1.746746358241776e-05,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/max_abs":0.2578125,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/norm":5.46875,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/mean":0.0004177093505859375,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_10/param/max_abs":1,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/norm":3.234375,"train/train/tensor_act_model_layers_61_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/mean":1.197986421175301e-08,"train/train/tensor_act_model_layers_81_mlp/std":0.22509820200032668,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/norm":0.007306602272967013,"train/train/tensor_act_model_layers_87_mlp/max_abs":4.15625,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/std":0.037109375,"train/train/tensor_act_model_layers_53_post_attention_layernorm/std":1.0000002142041693,"train/train/tensor_act_model_layers_39/frac_near_user_limit":0,"train/train/layer__model_layers_68/param/mean":0.0013301093567180187,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/max_abs":0.12158203125,"train/train/tensor_act_model_layers_3_post_attention_layernorm/mean":0.018280029296875,"train/train/tensor_act_model_layers_76_mlp_down_proj/mean":0.0020389556884765625,"train/train/tensor_act_model_layers_14_self_attn/max_abs":0.6953125,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/norm":3.890625,"train/train/layer_model_layers_12/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_q_proj/norm":7870.6566521053355,"train/train/tensor_act_model_layers_11_self_attn_q_proj/max_abs":5.8125,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/norm":5.25,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/std":0.04443359375,"train/train/tensor_act_model_layers_80_self_attn_q_proj/max_abs":5.9375,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/mean":-8.629285730421543e-08,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/mean":4.6253204345703125e-05,"train/train/tensor_param_model_layers_25_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/mean":-6.29425048828125e-05,"train/train/tensor_act_model_layers_48_self_attn_v_proj/mean":-0.00299072265625,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn_k_proj/max_abs":4.25,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/std":3.4939522471880625e-05,"train/train/tensor_act_model_layers_79_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/norm":0.00038896469524267213,"train/train/tensor_act_model_layers_60_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/mean":2.6941299438476562e-05,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/norm":0.01061112472589201,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/std":0.04248046875,"train/train/tensor_act_/norm":4.2136286624931545,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_19_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/std":9.420643366093428e-05,"train/train/tensor_act_model_layers_51_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/max_abs":0.001708984375,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/norm":3.234375,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/std":0.052001953125,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/std":0.02978515625,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/max_abs":0.2314453125,"train/train/tensor_act_model_layers_83_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/norm":7.6875,"train/train/tensor_act_model_layers_32_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/std":3.50447101825557e-05,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn_q_proj/norm":5711.079516777134,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_77/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_q_proj/max_abs":5.71875,"train/train/tensor_act_model_layers_90_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_46/act/mean":-0.006262559157151442,"train/train/tensor_param_model_layers_33_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_80/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/max_abs":0.00045013427734375,"train/train/layer__model_layers_82/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_self_attn_q_proj/std":0.8271504089107209,"train/train/layer_model_layers_17/grad/mean":-4.917525761875552e-07,"train/train/tensor_act_model_layers_65/std":1.4140739374593931,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/mean":-0.00015926361083984375,"train/train/tensor_act_model_layers_38_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/norm":0.0005821601735689731,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/mean":-0.00010967254638671875,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/max_abs":0.0002460479736328125,"train/train/tensor_param_model_layers_0_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_8_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_q_proj/mean":0.0706787109375,"train/train/tensor_act_model_layers_89_self_attn/mean":-0.01202392578125,"train/train/layer_model_layers_66/grad/max_abs":0.00128936767578125,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_39/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/mean":2.50060111284256e-06,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/max_abs":0.208984375,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/max_abs":0.00069427490234375,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_o_proj/max_abs":0.640625,"train/train/layer_model_layers_8/act/norm":16383.449037665092,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/mean":6.287882570177317e-08,"train/train/tensor_act_model_layers_25_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_15/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/act/mean":-0.021967227642352764,"train/train/tensor_act_model_layers_61_self_attn_k_proj/std":0.8710939980408182,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/norm":0.0009581878664976664,"train/train/tensor_act_model_layers_11_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/mean":-4.377216100692749e-08,"train/train/layer_model_layers_20/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/max_abs":0.000659942626953125,"train/train/tensor_act_model_layers_72_self_attn_v_proj/max_abs":2.46875,"train/train/tensor_act_model_layers_86_self_attn/std":0.17529367442388719,"train/train/tensor_act_model_layers_53_mlp/std":0.08947778951549998,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/norm":6.875,"train/train/tensor_act_model_layers_13_mlp_down_proj/mean":0.003505706787109375,"train/train/tensor_act_model_layers_51_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_post_attention_layernorm/std":1.0000000156869644,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/max_abs":1,"train/train/layer__model_layers_20/param/norm":19.45532743556376,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/mean":-0.000644683837890625,"train/train/tensor_act_model_layers_3_mlp_up_proj/frac_near_user_limit":0,"train/train/layer_model_layers_14/grad/mean":-3.599739733752148e-07,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/max_abs":0.1083984375,"train/train/tensor_act_model_layers_37_self_attn_k_proj/norm":5223.515564415172,"train/train/tensor_act_model_layers_48_self_attn/std":0.07019165411171444,"train/train/tensor_act_model_layers_27_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_46/grad/max_abs":0.00115966796875,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/norm":5.21875,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/max_abs":0.00011920928955078125,"train/train/layer__model_layers_9/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/mean":3.8067810237407684e-08,"train/train/layer__model_layers_53/param/norm":21.44722925177341,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/max_abs":9.393692016601562e-05,"train/train/tensor_act_model_layers_55_self_attn_v_proj/mean":-0.0057830810546875,"train/train/tensor_act_model_layers_79_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_up_proj/mean":-0.05963134765625,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/mean":-0.0003299713134765625,"train/train/tensor_act_model_layers_1/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/norm":0.00438089373945414,"train/train/tensor_act_model_layers_63_self_attn_k_proj/max_abs":5.90625,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/norm":4.03125,"train/train/tensor_act_model_layers_48_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/std":1.3366173835649545e-05,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/norm":0.010515818966781335,"train/train/layer_model_layers_68/grad/std":9.553376101825168e-05,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/std":1.7738735589334875e-05,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/std":1.952412915943858e-05,"train/train/tensor_act_model_layers_57_mlp_up_proj/max_abs":3.375,"train/train/tensor_act_model_layers_48/mean":0.070556640625,"train/train/tensor_act_model_layers_8_self_attn_v_proj/max_abs":1.5859375,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_52/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/max_abs":0.0003490447998046875,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp/mean":0.00293731689453125,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/max_abs":0.1220703125,"train/train/tensor_act_model_layers_74_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn/norm":1016.720683611927,"train/train/tensor_act_model_layers_44/std":1.302750663140772,"train/train/tensor_act_model_layers_90_post_attention_layernorm/max_abs":5.59375,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/max_abs":0.1533203125,"train/train/tensor_act_model_layers_73/norm":8747.96503189058,"train/train/layer_model_layers_64/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/mean":-2.156198024749756e-05,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/std":0.031494140625,"train/train/layer_model_layers_77/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/std":0.0284423828125,"train/train/tensor_act_model_layers_75/mean":0.07568359375,"train/train/tensor_act_model_layers_56_self_attn_q_proj/std":0.9033225545017668,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/mean":3.504753112792969e-05,"train/train/tensor_act_model_layers_36_self_attn_v_proj/std":0.3725596326504792,"train/train/tensor_act_model_layers_0_mlp_up_proj/mean":-0.04217529296875,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_input_layernorm/norm":5792.604858402403,"train/train/tensor_act_model_layers_41/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_v_proj/std":0.5556667187631227,"train/train/layer_model_layers_17/act/max_abs":18,"train/train/layer__model_layers_72/param/mean":0.0015960550531396256,"train/train/tensor_act_model_layers_53_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp_down_proj/std":0.11377005852924488,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/norm":0.023647322810870598,"train/train/tensor_act_model_layers_26_self_attn_v_proj/std":0.2836927341878522,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/norm":0.0008173294907386022,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_post_attention_layernorm/max_abs":5.375,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/norm":0.02411668868517535,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/max_abs":0.16015625,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/norm":5.09375,"train/train/layer__model_layers_53/param/max_abs":1,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/mean":-0.00017070770263671875,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/std":2.3285140690246808e-05,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp_up_proj/norm":3925.8275899566784,"train/train/tensor_act_model_layers_30_mlp_down_proj/std":0.04541049144477597,"train/train/tensor_act_model_layers_0_self_attn_q_proj/std":0.47412395691277226,"train/train/layer_model_layers_11/act/max_abs":18.25,"train/train/tensor_act_model_layers_9_self_attn/max_abs":1.5546875,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/norm":0.005218104296844347,"train/train/tensor_act_model_layers_43_self_attn_o_proj/norm":636.5367902032095,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/max_abs":0.0003833770751953125,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/norm":0.0016150637153638115,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_mlp_up_proj/frac_near_user_limit":0,"train/train/layer__model_layers_82/param/norm":23.77042707073224,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/norm":3.28125,"train/train/layer__model_layers_86/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_48_post_attention_layernorm/std":0.9960938098383867,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp/std":0.11938502695878023,"train/train/tensor_act_model_layers_13_self_attn_o_proj/std":0.03698751127455635,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/mean":-5.030632019042969e-05,"train/train/tensor_act_model_layers_8_self_attn_k_proj/std":1.2421875284902701,"train/train/layer_model_layers_22/grad/std":4.713009157030259e-05,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_q_proj/mean":-0.024566650390625,"train/train/tensor_act_model_layers_68_self_attn_q_proj/norm":5678.552870141603,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/mean":-2.0652078092098236e-07,"train/train/tensor_act_model_layers_52_self_attn_o_proj/std":0.053658940772566284,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_post_attention_layernorm/std":1.0000000055879354,"train/train/tensor_act_model_layers_33_post_attention_layernorm/norm":5792.60791016122,"train/train/tensor_act_model_layers_92_self_attn_v_proj/max_abs":3.203125,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/max_abs":0.1796875,"train/train/tensor_act_model_layers_83/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/mean":5.7079887483268976e-08,"train/train/tensor_act_model_layers_31_self_attn_q_proj/mean":-0.033203125,"train/train/layer_model_layers_73/grad/norm":0.05842739025748376,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/max_abs":0.142578125,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/norm":3.0625,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/std":0.04541015625,"train/train/layer_model_layers_28/grad/frac_near_user_limit":0,"train/train/layer_model_layers_12/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/mean":-1.737847924232483e-06,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/norm":4.28125,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42/norm":7509.931702162529,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/max_abs":0.142578125,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/std":1.909853216354975e-05,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/std":0.02587890625,"train/train/tensor_act_model_layers_19_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_90/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_mlp/std":0.1596702173057205,"train/train/tensor_act_model_layers_3_post_attention_layernorm/std":1.0000000237487252,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/std":0.059326171875,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/mean":-0.000518798828125,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/norm":5.65625,"train/train/tensor_act_model_layers_58_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/max_abs":0.0014495849609375,"train/train/tensor_act_model_layers_41_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_10_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_k_proj/norm":5110.877535400742,"train/train/tensor_act_model_layers_3_self_attn_o_proj/norm":721.508086273818,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/mean":6.405753083527088e-08,"train/train/layer_model_layers_47/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/std":5.293691839522257e-05,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/max_abs":0.2041015625,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/max_abs":0.125,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_mlp_down_proj/mean":0.00628662109375,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/mean":-6.580352783203125e-05,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/mean":-1.7692218534648418e-07,"train/train/tensor_act_model_layers_70_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_post_attention_layernorm/std":1.0000000192303558,"train/train/layer_model_layers_55/act/std":0.700286885409109,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/norm":5.625,"train/train/tensor_act_model_layers_78_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37/norm":7593.038863149167,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/std":2.8973217177205963e-05,"train/train/tensor_act_model_layers_5_input_layernorm/norm":5792.606689455929,"train/train/tensor_param_model_layers_82_input_layernorm_weight/mean":1,"train/train/global/param/max_abs":1,"train/train/layer__model_layers_85/param/max_abs":1,"train/train/tensor_act_model_layers_63_input_layernorm/norm":5792.6143798835055,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/max_abs":7.295608520507812e-05,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/mean":-1.1585652828216553e-06,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/norm":5.21875,"train/train/tensor_act_model_layers_18_self_attn_o_proj/std":0.06152458013198962,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/max_abs":0.1962890625,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_17_mlp_down_proj/std":0.05462658312054835,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/mean":-1.2776581570506096e-08,"train/train/tensor_act_model_layers_90_self_attn/std":0.1552814558243816,"train/train/tensor_act_model_layers_83_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_k_proj/std":0.81152525533204,"train/train/tensor_act_model_layers_89_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/norm":2.8125,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/norm":3.78125,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/norm":0.003554880338720045,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/std":4.74469437271157e-05,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/norm":0.0013621941559227048,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/max_abs":0.0009765625,"train/train/tensor_act_model_layers_13_post_attention_layernorm/mean":0.0011720657348632812,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_input_layernorm_weight/std":0,"train/train/layer_model_layers_6/grad/std":5.040407267559093e-05,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/mean":-0.00015163421630859375,"train/train/tensor_act_model_layers_87_post_attention_layernorm/std":0.9960953880745378,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/mean":2.483429852873087e-07,"train/train/tensor_act_model_layers_88_self_attn_k_proj/mean":0.04913330078125,"train/train/tensor_act_model_layers_14/std":1.4023710060758259,"train/train/tensor_act_model_layers_35_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_input_layernorm/std":1.000002210957348,"train/train/tensor_act_model_layers_16_input_layernorm/std":1.0000000478466962,"train/train/tensor_act_model_layers_37_mlp_up_proj/max_abs":3.359375,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/max_abs":0.0001811981201171875,"train/train/tensor_act_model_layers_13_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/mean":-7.136259227991104e-08,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23/max_abs":17.75,"train/train/tensor_act_model_layers_48_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/norm":0.0008331621079972148,"train/train/tensor_act_model_layers_39_mlp_down_proj/norm":311.1432610160008,"train/train/tensor_act_model_layers_10/mean":-0.00510406494140625,"train/train/tensor_act_model_layers_27_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/norm":6.28125,"train/train/tensor_act_model_layers_83_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_44/act/norm":14106.98138353816,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/norm":6.25,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/std":3.119905708855915e-05,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/max_abs":0.0001621246337890625,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/max_abs":0.12060546875,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/mean":0.0005950927734375,"train/train/tensor_act_model_layers_38_post_attention_layernorm/norm":5792.608398438374,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/max_abs":0.12451171875,"train/train/tensor_act_model_layers_81_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_4_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/max_abs":0.126953125,"train/train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_post_attention_layernorm/std":0.996095754583997,"train/train/tensor_act_model_layers_83_input_layernorm/max_abs":5.4375,"train/train/tensor_act_model_layers_33_mlp_down_proj/norm":297.5330767104795,"train/train/tensor_act_model_layers_11_self_attn_k_proj/std":1.0156250880314714,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/max_abs":0.146484375,"train/train/tensor_act_model_layers_4_mlp/mean":-0.01123046875,"train/train/tensor_act_model_layers_61_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59/norm":7722.115188110267,"train/train/tensor_act_model_layers_11_post_attention_layernorm/norm":5792.60791016089,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/max_abs":0.103515625,"train/train/tensor_param_model_layers_91_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/max_abs":0.166015625,"train/train/tensor_act_model_layers_64_post_attention_layernorm/mean":0.0531005859375,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/norm":4.5,"train/train/tensor_act_model_layers_42_mlp/std":0.06250000217551129,"train/train/tensor_act_model_layers_1_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/max_abs":0.00037384033203125,"train/train/tensor_act_model_layers_17/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/std":0.046142578125,"train/train/tensor_act_model_layers_14_mlp_down_proj/mean":0.0042266845703125,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_input_layernorm/mean":0.03411865234375,"train/train/tensor_param_model_layers_36_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_77_self_attn/std":0.1606485554953938,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/mean":1.3969838619232178e-09,"train/train/tensor_act_model_layers_68_input_layernorm/std":1.0000008996572023,"train/train/layer_model_layers_28/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_up_proj/max_abs":2.21875,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/mean":6.482878234237432e-08,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/max_abs":0.00078582763671875,"train/train/layer_model_layers_60/grad/mean":-4.874943361825393e-07,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_46/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/mean":5.53131103515625e-05,"train/train/tensor_act_model_layers_19/mean":0.0325927734375,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/norm":0.015011949994512028,"train/train/tensor_act_model_layers_92_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/max_abs":0.142578125,"train/train/tensor_param_model_layers_57_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_76_self_attn_v_proj/max_abs":2.625,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_54/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_64/grad/max_abs":0.00109100341796875,"train/train/tensor_param_model_layers_44_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/norm":0.01246974532139189,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/mean":8.42846930027008e-08,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_post_attention_layernorm/norm":5792.606567388342,"train/train/tensor_act_model_layers_71_mlp/norm":885.6468437714266,"train/train/tensor_act_model_layers_72/norm":8695.030314094403,"train/train/tensor_act_model_layers_91_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_post_attention_layernorm/max_abs":6.03125,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/std":3.3461592455013824e-05,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/mean":-2.8312206268310547e-05,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/max_abs":0.000896453857421875,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/max_abs":0.00408935546875,"train/train/tensor_act_model_layers_41_mlp/mean":0.00390625,"train/train/tensor_act_model_layers_68_post_attention_layernorm/norm":5792.608886720242,"train/train/layer__model_layers_38/param/mean":0.0013900971077905617,"train/train/tensor_act_model_layers_16_post_attention_layernorm/norm":5792.605957031507,"train/train/tensor_act_model_layers_74_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_75/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/max_abs":6.580352783203125e-05,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_post_attention_layernorm/std":1.0000000372529023,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_v_proj/norm":1611.1583766355118,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/norm":0.011923550126867664,"train/train/tensor_act_model_layers_80_self_attn_q_proj/norm":7052.189965398168,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/mean":7.348135113716125e-07,"train/train/tensor_act_model_layers_13_mlp_up_proj/mean":-0.0557861328125,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/norm":0.0076158632447799425,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_89/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/std":2.399295490563542e-05,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_93_mlp/std":1.4473013879885055,"train/train/tensor_act_model_layers_40_self_attn_q_proj/std":1.150395702616394,"train/train/tensor_param_model_layers_36_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/std":0.044189453125,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/std":0.039794921875,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/mean":0.0004119873046875,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/max_abs":8.630752563476562e-05,"train/train/tensor_act_model_layers_17_mlp/std":0.05462658312054835,"train/train/tensor_act_model_layers_51_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/norm":5.0625,"train/train/tensor_act_model_layers_58_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp_up_proj/mean":-0.078369140625,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/max_abs":0.00091552734375,"train/train/layer_model_layers_9/act/norm":18027.15163582445,"train/train/tensor_act_model_layers_49_self_attn_v_proj/norm":2198.1303237784573,"train/train/tensor_act_model_layers_80_self_attn/std":0.23828528167658544,"train/train/tensor_act_model_layers_0_mlp_up_proj/max_abs":5.59375,"train/train/tensor_act_model_layers_83_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_post_attention_layernorm/max_abs":6.125,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/max_abs":0.00010442733764648438,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/std":3.591162309654148e-05,"train/train/tensor_act_model_layers_61_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91/max_abs":15.75,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/std":1.860480344015998e-05,"train/train/tensor_act_model_layers_87_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26/max_abs":17.75,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/max_abs":0.0003566741943359375,"train/train/layer__model_layers_27/param/mean":0.0015887708262236739,"train/train/tensor_act_model_layers_31_self_attn_k_proj/std":0.7968750116871852,"train/train/tensor_act_model_layers_73_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/max_abs":0.255859375,"train/train/tensor_act_model_layers_6_post_attention_layernorm/norm":5792.605102548954,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/mean":4.384201020002365e-07,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/max_abs":0.00024318695068359375,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/std":6.664212061145018e-05,"train/train/tensor_act_model_layers_37_mlp_down_proj/max_abs":0.5390625,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_52/param/max_abs":1,"train/train/tensor_act_model_layers_92_input_layernorm/std":1.000001361592677,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_12/grad/mean":-3.2664468445104665e-07,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/mean":-2.5331974029541016e-06,"train/train/tensor_act_model_layers_88_post_attention_layernorm/norm":5792.603759773553,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/mean":3.886222839355469e-05,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/mean":0.00011873245239257812,"train/train/layer_model_layers_37/grad/max_abs":0.00168609619140625,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/norm":0.018863517135690393,"train/train/tensor_act_model_layers_33_mlp/std":0.051330680973195014,"train/train/tensor_act_model_layers_88_mlp_down_proj/mean":-0.00015461444854736328,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/norm":6.3125,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/max_abs":0.224609375,"train/train/layer__model_layers_15/param/std":0.0491240825522255,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/std":0.041015625,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/std":0.052001953125,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/std":1.4720986127560155e-05,"train/train/tensor_act_model_layers_70_self_attn_q_proj/mean":-0.0025310516357421875,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/std":0,"train/train/layer__model_layers_19/param/norm":19.327998685925166,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/std":1.4922741401368235e-05,"train/train/layer_model_layers_11/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_26/grad/std":3.681218514170972e-05,"train/train/tensor_act_model_layers_6_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/std":0.0303955078125,"train/train/layer_model_layers_19/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/mean":-3.310851752758026e-07,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/norm":0.02957498158287449,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/mean":6.914138793945312e-05,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/max_abs":0.0004405975341796875,"train/train/tensor_act_model_layers_38_self_attn_k_proj/mean":0.0147705078125,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/std":0.0240478515625,"train/train/tensor_act_model_layers_64/std":1.388676065258888,"train/train/tensor_act_model_layers_46_post_attention_layernorm/max_abs":5.6875,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_mlp/mean":0.0088348388671875,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_mlp_down_proj/max_abs":0.39453125,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/std":0.022216796875,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_2/act/norm":15870.7596089345,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/mean":-5.052424967288971e-07,"train/train/tensor_act_model_layers_65_self_attn_k_proj/norm":4454.863416401022,"train/train/tensor_act_model_layers_9_mlp_up_proj/std":0.21191409765849722,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/mean":-7.904600352048874e-08,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_38/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/norm":0.029353087855960976,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/max_abs":0.1376953125,"train/train/layer_model_layers_3/grad/norm":0.05295488102558196,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/mean":-0.0003185272216796875,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/max_abs":0.1455078125,"train/train/tensor_act_model_layers_6_self_attn_k_proj/max_abs":6.375,"train/train/tensor_act_model_layers_80_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_post_attention_layernorm/mean":0.05517578125,"train/train/tensor_act_model_layers_88_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/max_abs":0.126953125,"train/train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_v_proj/std":0.486328156597642,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/std":1.551597296583743e-05,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/norm":5.6875,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_input_layernorm/mean":0.04095458984375,"train/train/tensor_param_model_embed_tokens_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/max_abs":0.0002880096435546875,"train/train/tensor_act_model_layers_62/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/mean":-1.1961674317717552e-08,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/norm":5.78125,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/max_abs":0.00023555755615234375,"train/train/tensor_act_model_layers_17_self_attn_k_proj/max_abs":4.53125,"train/train/tensor_act_model_layers_86_self_attn_k_proj/mean":0.1143798828125,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/max_abs":0.000667572021484375,"train/train/layer_model_layers_90/act/norm":19725.04895605531,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/max_abs":0.00019931793212890625,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/std":6.917215127091959e-05,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/std":0.0306396484375,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_k_proj/max_abs":5.53125,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/norm":0.16427017872965577,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/max_abs":0.000499725341796875,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/norm":0.0028513324657829095,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/max_abs":0.000640869140625,"train/train/tensor_param_model_layers_55_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_46_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_75/param/norm":23.416031834514573,"train/train/tensor_act_model_layers_25_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_57/param/norm":21.927147664197,"train/train/tensor_act_model_layers_19_self_attn/std":0.024506068810946684,"train/train/tensor_act_model_layers_23_mlp/norm":232.44342835376997,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/norm":3.96875,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/mean":2.532033249735832e-07,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/max_abs":0.0007171630859375,"train/train/tensor_param_model_layers_4_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_56_self_attn_v_proj/mean":0.0039825439453125,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/mean":5.45009970664978e-06,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/max_abs":0.00052642822265625,"train/train/tensor_act_model_layers_41_post_attention_layernorm/norm":5792.60546875464,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/norm":6.5625,"train/train/tensor_act_model_layers_4_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/mean":-4.343688488006592e-06,"train/train/tensor_act_model_layers_42_self_attn_o_proj/norm":393.05543529869465,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/std":6.0408626741551705e-05,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/mean":-9.620562195777893e-07,"train/train/tensor_param_model_layers_39_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_3_input_layernorm/std":1.0000000051222742,"train/train/tensor_act_model_layers_25_input_layernorm/mean":0.0340576171875,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/std":0.033203125,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/max_abs":0.138671875,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/mean":-1.0014045983552933e-06,"train/train/layer__model_layers_64/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/max_abs":0.11962890625,"train/train/tensor_act_model_layers_15_mlp_up_proj/max_abs":2.15625,"train/train/tensor_act_model_layers_58_input_layernorm/norm":5792.61120605781,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_mlp_up_proj/max_abs":3.390625,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_post_attention_layernorm/mean":0.05841064453125,"train/train/tensor_param_model_layers_34_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_85_mlp_up_proj/mean":-0.137451171875,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/max_abs":0.2060546875,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/norm":0.0013676634028350234,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/std":2.9672866123667117e-05,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/norm":0.027113833431836404,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/max_abs":0.11669921875,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/norm":5.0625,"train/train/tensor_act_model_layers_19_self_attn_v_proj/mean":-0.003787994384765625,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/std":0.02197265625,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/norm":0.013079586195812612,"train/train/layer__model_layers_53/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_34/param/norm":20.559679945298882,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/mean":-0.0006103515625,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/max_abs":0.000843048095703125,"train/train/tensor_act_model_layers_64_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_45_post_attention_layernorm/norm":5792.61291504097,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/mean":-4.260800778865814e-08,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/mean":-8.058547973632812e-05,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/mean":1.0337680578231812e-06,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/mean":5.760230123996735e-07,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/std":0.04345703125,"train/train/tensor_act_model_layers_74/max_abs":12.9375,"train/train/tensor_param_model_layers_39_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/mean":4.1961669921875e-05,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/mean":1.5169382095336914e-05,"train/train/tensor_act_model_layers_87/mean":0.1669921875,"train/train/tensor_act_model_layers_4_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65/mean":0.06060791015625,"train/train/tensor_act_model_layers_59_self_attn_k_proj/std":0.7695312984098622,"train/train/tensor_param_model_layers_9_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/mean":-2.9685907065868378e-09,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/mean":0.00021839141845703125,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/std":1.5181512706115416e-05,"train/train/tensor_act_model_layers_72_mlp_down_proj/norm":952.6632826005591,"train/train/tensor_act_model_layers_55_mlp_down_proj/norm":564.1104980056358,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/norm":0.02835416414922796,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/mean":0.0004367828369140625,"train/train/tensor_act_model_layers_55_mlp/norm":564.1104980056358,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/std":0.031982421875,"train/train/tensor_act_model_layers_82_self_attn_v_proj/mean":0.0006042215973138809,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/mean":-2.0442530512809753e-07,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/norm":5.78125,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/norm":5.34375,"train/train/tensor_act_model_layers_62_self_attn/norm":879.6333991944758,"train/train/tensor_act_model_layers_17_self_attn_v_proj/std":0.24121100009090174,"train/train/layer_model_layers_66/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_k_proj/mean":0.105224609375,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/max_abs":0.1181640625,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/mean":-0.001129150390625,"train/train/tensor_act_model_layers_80_mlp/std":0.18457033063368733,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/std":0.052490234375,"train/train/tensor_act_model_layers_36_mlp/mean":0.00257110595703125,"train/train/tensor_act_model_layers_54/std":1.3183754870525144,"train/train/tensor_act_model_layers_36_input_layernorm/max_abs":6.125,"train/train/tensor_act_model_layers_38_mlp/mean":-0.0092620849609375,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/max_abs":0.1318359375,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/norm":0.016395751682289843,"train/train/layer__model_layers_52/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/norm":0.003976751916941989,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp_up_proj/std":0.4687505722042406,"train/train/layer_model_layers_36/act/norm":14062.167193092539,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/mean":0.00015544891357421875,"train/train/tensor_act_model_layers_56_self_attn_k_proj/norm":4496.517861534193,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/std":0.00014880263100909617,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_75/act/norm":15825.872573301787,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/mean":0.000705718994140625,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/mean":-1.0337680578231812e-07,"train/train/layer_model_layers_37/act/norm":14086.345909616226,"train/train/tensor_act_model_layers_77_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/max_abs":0.0001659393310546875,"train/train/tensor_act_model_layers_16_mlp_down_proj/mean":0.00293731689453125,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/max_abs":0.1376953125,"train/train/tensor_act_model_layers_64_self_attn_q_proj/mean":0.017364501953125,"train/train/tensor_act_model_layers_76_self_attn_q_proj/mean":-0.0545654296875,"train/train/tensor_param_model_layers_65_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_59/std":1.3320370894944513,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/std":1.98189221940211e-05,"train/train/layer_model_layers_66/act/norm":15206.349098774415,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/mean":-1.5914440155029297e-05,"train/train/tensor_act_model_layers_34/max_abs":17.875,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/std":0.00022735248517169386,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_lm_head/max_abs":16.125,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/mean":-0.00058746337890625,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_o_proj/mean":-0.00142669677734375,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_58/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/std":4.036460464840528e-05,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/max_abs":1,"train/train/layer__model_layers_43/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_82/act/mean":0.0015676624786395293,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/std":0.053955078125,"train/train/tensor_param_model_layers_84_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_84_self_attn_v_proj/max_abs":2.546875,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/std":2.6637253094624617e-05,"train/train/tensor_act_model_layers_80_mlp_down_proj/mean":0.00266265869140625,"train/train/tensor_param_model_layers_55_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/mean":-1.6512349247932434e-06,"train/train/layer_model_layers_6/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_q_proj/norm":6210.249650085276,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/norm":0.002596876002132484,"train/train/layer_model_layers_35/act/std":0.6704377585680107,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/std":9.880927515594978e-05,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/mean":-1.5506520867347717e-07,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/mean":2.0046718418598175e-07,"train/train/tensor_act_model_layers_27_mlp/mean":0.00447845458984375,"train/train/tensor_act_model_layers_53_self_attn_v_proj/std":0.361328431138186,"train/train/tensor_act_model_layers_17/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_v_proj/mean":-0.00035184621810913086,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/norm":8.25,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_26/param/max_abs":1,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/max_abs":0.126953125,"train/train/tensor_act_model_layers_72/mean":0.05291748046875,"train/train/layer_model_layers_25/act/std":0.6794937629947498,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/max_abs":0.19140625,"train/train/tensor_act_model_layers_26_post_attention_layernorm/norm":5792.601806641289,"train/train/tensor_act_model_layers_87_mlp_up_proj/std":0.6445321285357413,"train/train/tensor_act_model_layers_23_self_attn_q_proj/max_abs":5.9375,"train/train/tensor_act_model_layers_45_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/max_abs":6.389617919921875e-05,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/mean":-1.5737023204565048e-06,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_10_self_attn_v_proj/max_abs":1.7578125,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/norm":6.40625,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/max_abs":0.00012493133544921875,"train/train/tensor_act_model_layers_90_self_attn_v_proj/max_abs":4.0625,"train/train/tensor_act_model_layers_8_self_attn_v_proj/std":0.2773438040429385,"train/train/layer__model_layers_75/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_44_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp/max_abs":1.3359375,"train/train/tensor_act_model_layers_31_input_layernorm/norm":5792.61315918245,"train/train/tensor_act_model_layers_84_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_post_attention_layernorm/mean":0.03118896484375,"train/train/layer_model_layers_18/grad/max_abs":0.0020751953125,"train/train/tensor_act_model_layers_25_self_attn_o_proj/norm":334.2355866445592,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/max_abs":0.0001850128173828125,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/max_abs":0.2109375,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/norm":5.75,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/max_abs":0.1396484375,"train/train/tensor_act_model_layers_93_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/std":1.4224687913020387e-05,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn/mean":0.0008821487426757812,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/max_abs":0.00067138671875,"train/train/tensor_act_model_layers_35_self_attn_q_proj/max_abs":6.03125,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/std":2.849671914833911e-05,"train/train/tensor_act_model_layers_81_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_post_attention_layernorm/std":0.9960939519545406,"train/train/layer_model_layers_49/act/norm":13984.720369380466,"train/train/tensor_act_model_layers_81_mlp_up_proj/mean":-0.12939453125,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/max_abs":0.158203125,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/mean":-2.4330802261829376e-07,"train/train/tensor_act_model_layers_65_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/norm":3.78125,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/std":1.728353271743285e-05,"train/train/tensor_act_model_layers_55_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/norm":0.03320626636679196,"train/train/tensor_act_model_layers_86_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/std":2.4999564012692354e-05,"train/train/tensor_act_model_layers_30_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/mean":3.4440308809280396e-06,"train/train/tensor_act_model_layers_33_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40/norm":7559.4174278272485,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_post_attention_layernorm/norm":5792.608276379138,"train/train/tensor_act_model_layers_29_self_attn_q_proj/std":0.9687502672594994,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/max_abs":0.0020751953125,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/max_abs":0.0009002685546875,"train/train/tensor_act_model_layers_88_self_attn_o_proj/mean":-0.008880615234375,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_k_proj/std":0.7509890110090323,"train/train/tensor_act_model_layers_21_self_attn/mean":-0.0012874603271484375,"train/train/layer_model_layers_85/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/norm":0.00423708987960839,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/mean":-1.307344064116478e-07,"train/train/tensor_act_model_layers_81_mlp_down_proj/norm":1303.9379899180044,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/norm":7.59375,"train/train/tensor_act_model_layers_93/max_abs":30,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/mean":9.73232090473175e-08,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/std":0.038330078125,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/max_abs":0.000537872314453125,"train/train/tensor_act_model_layers_55_self_attn/mean":-0.002490997314453125,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/norm":0.007566128678909749,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/max_abs":0.00151824951171875,"train/train/tensor_act_model_layers_71_self_attn_k_proj/max_abs":5.1875,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/max_abs":0.154296875,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/mean":0.00015163421630859375,"train/train/layer_model_layers_7/grad/norm":0.05498962537437466,"train/train/tensor_act_model_layers_21_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/std":6.955225617401524e-05,"train/train/layer_model_layers_45/act/max_abs":16.375,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/norm":0.010506958838179342,"train/train/tensor_act_model_layers_86_input_layernorm/mean":0.0782470703125,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/max_abs":0.1630859375,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/max_abs":0.0007171630859375,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/std":0.02783203125,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/mean":7.655471563339233e-07,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/std":0.030029296875,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/std":0.035400390625,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/max_abs":0.0002765655517578125,"train/train/tensor_act_model_layers_79_input_layernorm/norm":5792.610229494157,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/norm":0.054019251155768465,"train/train/tensor_act_model_layers_38_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/norm":0.018589689468541948,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/max_abs":0.216796875,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_75/grad/std":8.788884965651765e-05,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/std":8.791410665764071e-05,"train/train/tensor_act_model_layers_75_self_attn_o_proj/std":0.22168658800875377,"train/train/layer_model_layers_44/act/mean":-0.007597446441650391,"train/train/tensor_act_model_layers_17_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_v_proj/std":0.507812677209163,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp_up_proj/mean":-0.09716796875,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/max_abs":0.1181640625,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn_k_proj/max_abs":5.15625,"train/train/layer__model_layers_85/param/mean":0.0012499352513163026,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn_o_proj/norm":787.1452617548202,"train/train/layer_model_layers_85/grad/max_abs":0.00101470947265625,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/std":0.0272216796875,"train/train/layer__model_layers_74/param/std":0.056337339764908075,"train/train/tensor_act_model_layers_34_mlp_down_proj/mean":0.0080413818359375,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/std":8.046898381143892e-05,"train/train/layer__model_layers_4/param/max_abs":1,"train/train/tensor_act_model_layers_7_mlp/max_abs":0.6015625,"train/train/tensor_act_model_layers_76_self_attn/max_abs":1.859375,"train/train/tensor_act_model_layers_13_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_66_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp_down_proj/std":0.31347809996183323,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/mean":1.8747523427009583e-06,"train/train/tensor_param_model_layers_56_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_61/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/norm":0.00041263031230728137,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/std":7.06649234307837e-05,"train/train/layer__model_layers_48/param/std":0.052959456171684814,"train/train/layer_model_layers_31/act/norm":13771.264577631771,"train/train/tensor_act_model_layers_8_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn_o_proj/norm":1387.4049382221137,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/norm":2.796875,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/norm":0.0008621490102810288,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/max_abs":0.1357421875,"train/train/tensor_param_model_layers_12_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_51_input_layernorm/norm":5792.602416993272,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_77_mlp_down_proj/max_abs":1.515625,"train/train/tensor_param_model_layers_68_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn_k_proj/std":0.7695319858295776,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/max_abs":0.00069427490234375,"train/train/layer_model_layers_30/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_q_proj/mean":0.0614013671875,"train/train/tensor_param_model_layers_20_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_41_mlp_up_proj/norm":3282.107122185266,"train/train/tensor_act_model_layers_71_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/std":0.050048828125,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26/mean":0.0491943359375,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/std":1.3259721588626743e-05,"train/train/tensor_act_model_layers_74_mlp/max_abs":1.703125,"train/train/tensor_act_model_layers_77_input_layernorm/max_abs":5.78125,"train/train/layer_model_layers_75/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/max_abs":0.0007171630859375,"train/train/tensor_act_model_layers_59_mlp_down_proj/mean":0.017669677734375,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/mean":0.0006866455078125,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_mlp_up_proj/std":0.30078184449768763,"train/train/tensor_act_model_layers_64_mlp/norm":699.9703697549551,"train/train/tensor_act_model_layers_17_mlp_up_proj/norm":2354.4620342400276,"train/train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn/norm":514.2676470185705,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/max_abs":0.1005859375,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/norm":0.03890061211239241,"train/train/tensor_param_model_layers_77_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_35_mlp_down_proj/std":0.04870621243788334,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/std":4.694345822431742e-05,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/max_abs":0.0006103515625,"train/train/tensor_act_model_layers_2/norm":8735.331201771394,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/mean":1.2747477740049362e-07,"train/train/tensor_act_model_layers_56_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/mean":-0.000629425048828125,"train/train/tensor_act_model_layers_0/mean":0.04302978515625,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/norm":0.0013037170899349586,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/mean":2.8014183044433594e-05,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/norm":0.005492061928280808,"train/train/tensor_act_model_layers_8_post_attention_layernorm/max_abs":5.65625,"train/train/tensor_act_model_layers_7_self_attn_v_proj/mean":0.012969970703125,"train/train/tensor_act_model_layers_69_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/norm":0.02182730174713752,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_post_attention_layernorm/norm":5792.611694338925,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/max_abs":0.1865234375,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_32/norm":7573.514331450654,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/std":0.046875,"train/train/tensor_act_model_layers_66_self_attn_o_proj/norm":1192.525413421562,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/std":0.0233154296875,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/mean":1,"train/train/layer__model_layers_16/param/mean":0.001694582553809965,"train/train/tensor_act_model_layers_36_self_attn_v_proj/mean":-0.00638580322265625,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/std":0.00014722532819426253,"train/train/tensor_act_model_layers_1_post_attention_layernorm/max_abs":4.5,"train/train/tensor_param_model_layers_44_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_9_self_attn_o_proj/max_abs":1.5546875,"train/train/tensor_act_model_layers_10_self_attn/std":0.06774949313121226,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_42/grad/std":5.0202461946102096e-05,"train/train/tensor_act_model_layers_3_self_attn_q_proj/mean":-0.05712890625,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_32/act/max_abs":17.75,"train/train/tensor_act_model_layers_21_post_attention_layernorm/max_abs":6.03125,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/mean":0.000217437744140625,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/std":1.5287212493280195e-05,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/mean":-6.940215826034546e-06,"train/train/tensor_act_model_layers_89_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_16/act/mean":-0.013549071091871995,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/norm":0.005236277519189276,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/std":0.033447265625,"train/train/tensor_act_model_layers_3_mlp_down_proj/std":0.08569370817184105,"train/train/tensor_act_model_layers_43_input_layernorm/max_abs":6.125,"train/train/tensor_param_model_layers_68_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/max_abs":0.0004673004150390625,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_q_proj/max_abs":5.3125,"train/train/tensor_act_model_layers_93_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_k_proj/max_abs":4.96875,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/mean":0.00021839141845703125,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/max_abs":0.00058746337890625,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/max_abs":0.000827789306640625,"train/train/tensor_act_model_layers_7_self_attn/mean":0.0003714561462402344,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/std":3.257778861075762e-05,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/mean":0.00038909912109375,"train/train/tensor_act_model_layers_86_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/std":1.9760997184553474e-05,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn_o_proj/max_abs":1.5546875,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/mean":-5.343463271856308e-08,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/std":0.0001251725167932728,"train/train/tensor_act_model_layers_85_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/std":0.0240478515625,"train/train/layer__model_layers_36/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/max_abs":0.21875,"train/train/layer_model_layers_87/grad/std":0.00011326873879118833,"train/train/tensor_act_model_layers_90_mlp_down_proj/std":0.5224637040394773,"train/train/tensor_act_model_layers_7/max_abs":18.5,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/norm":0.000391116940779478,"train/train/tensor_param_model_layers_80_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_23_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_up_proj/max_abs":3.875,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/std":5.997473357475748e-05,"train/train/tensor_act_model_layers_23_self_attn/mean":-0.0007476806640625,"train/train/layer_model_layers_58/grad/mean":-3.1839027512650037e-07,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/std":0.04541015625,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/max_abs":0.201171875,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/max_abs":0.0003147125244140625,"train/train/tensor_act_model_layers_35_self_attn_o_proj/max_abs":1.578125,"train/train/tensor_act_model_layers_61_self_attn_o_proj/std":0.09619530942408828,"train/train/tensor_act_model_layers_39_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/mean":3.598397597670555e-07,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/max_abs":0.25390625,"train/train/tensor_param_model_norm_weight/mean":1,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/std":2.0435295778513315e-05,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_mlp_up_proj/frac_near_user_limit":0,"train/train/layer__model_layers_37/param/norm":20.517552746444323,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/mean":1.2513191904872656e-07,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_66/grad/std":7.908935160734357e-05,"train/train/layer__model_layers_68/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/max_abs":0.00014591217041015625,"train/train/tensor_param_model_layers_28_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/std":0.042236328125,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_mlp_down_proj/max_abs":3.9375,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_70/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/max_abs":0.169921875,"train/train/layer_model_layers_71/act/mean":-0.011831939220428467,"train/train/tensor_act_model_layers_93_input_layernorm/mean":0.05279541015625,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/mean":2.6047229766845703e-05,"train/train/layer_model_layers_27/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/mean":-1.5446916222572327e-05,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/mean":-0.00067138671875,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/mean":-0.000293731689453125,"train/train/tensor_act_model_layers_86_self_attn_o_proj/mean":-0.0006361007690429688,"train/train/tensor_act_model_layers_50_self_attn_o_proj/norm":327.12346802635545,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/norm":5.5625,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/norm":0.00046377212403601436,"train/train/tensor_act_model_layers_30_post_attention_layernorm/mean":0.0560302734375,"train/train/tensor_act_model_layers_77_post_attention_layernorm/mean":0.05023193359375,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_post_attention_layernorm/mean":0.0556640625,"train/train/tensor_act_model_layers_78_self_attn/std":0.17285466178035025,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/mean":-5.452893674373627e-07,"train/train/tensor_act_model_layers_40_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/max_abs":0.000499725341796875,"train/train/tensor_act_model_layers_47_self_attn/norm":992.1523644585675,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/max_abs":0.000720977783203125,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66/max_abs":13.375,"train/train/tensor_act_model_layers_85_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp_down_proj/norm":283.46965340995047,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/mean":-2.60770320892334e-07,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/std":3.211273705035924e-05,"train/train/tensor_act_model_layers_46_self_attn_v_proj/norm":2135.7827905793283,"train/train/layer__model_layers_20/param/std":0.0480097812942865,"train/train/tensor_act_model_layers_35_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/max_abs":0.1318359375,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp_down_proj/norm":232.67016588797628,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/max_abs":0.259765625,"train/train/tensor_param_model_layers_0_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/max_abs":0.12451171875,"train/train/tensor_act_model_layers_26/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_34/act/mean":-0.0042987236609825724,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/std":0.0390625,"train/train/tensor_act_model_layers_85_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/mean":-3.711320459842682e-07,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn_k_proj/max_abs":5.0625,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_1/act/std":0.8065540640498908,"train/train/tensor_act_model_layers_80/norm":9998.784914501299,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/mean":0.00011777877807617188,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_post_attention_layernorm/norm":5792.612915040373,"train/train/layer_model_layers_24/grad/mean":-6.869324888342442e-07,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_74/grad/norm":0.06232370426714431,"train/train/tensor_act_model_layers_82_self_attn_k_proj/norm":4757.765172373195,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn_q_proj/std":0.9179734087886684,"train/train/tensor_act_model_layers_2_self_attn_q_proj/std":1.3066450879625455,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/std":6.981546700379647e-05,"train/train/tensor_act_model_layers_18_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn_o_proj/mean":-0.002857208251953125,"train/train/tensor_param_model_layers_28_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/std":4.2622731345750866e-05,"train/train/global/act/std":0.8820047890617934,"train/train/tensor_act_model_layers_0/max_abs":18,"train/train/tensor_act_model_layers_83/norm":10553.83583569514,"train/train/tensor_act_model_layers_48_mlp_up_proj/frac_near_user_limit":0,"train/train/layer_model_layers_33/act/std":0.6597317134209795,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/std":0.03662109375,"train/train/tensor_act_model_layers_8/mean":-0.0065155029296875,"train/train/tensor_act_model_layers_57_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_mlp/mean":0.0206298828125,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/max_abs":0.00113677978515625,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/std":0.03955078125,"train/train/layer__model_layers_28/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/std":9.764028691426513e-06,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/mean":-2.6337802410125732e-06,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/max_abs":0.00189208984375,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/mean":9.629875421524048e-07,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/max_abs":0.23828125,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/mean":0.0003070831298828125,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/norm":6.40625,"train/train/layer_model_layers_40/grad/max_abs":0.0019073486328125,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/mean":-6.198883056640625e-05,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/mean":-9.059906005859375e-05,"train/train/tensor_act_model_layers_19_self_attn_k_proj/norm":4622.449600139848,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_93/grad/mean":-8.076281169089438e-07,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/mean":3.2316893339157104e-07,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/max_abs":0.1943359375,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/max_abs":0.1337890625,"train/train/tensor_act_model_layers_49_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/mean":1,"train/train/layer__model_layers_84/param/max_abs":1,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/mean":0.00017261505126953125,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/mean":-0.0011444091796875,"train/train/tensor_act_model_layers_78_self_attn/mean":0.00043952465057373047,"train/train/tensor_act_model_layers_64_self_attn_k_proj/std":0.8457061556085638,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/std":6.557747527831865e-05,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/max_abs":0.001495361328125,"train/train/tensor_act_model_layers_71_mlp_down_proj/mean":-0.0011882781982421875,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/std":0.022216796875,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/norm":0.015663516731933235,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_embed_tokens_weight/max_abs":0.0089111328125,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_post_attention_layernorm/mean":0.039794921875,"train/train/tensor_act_model_layers_88_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12/max_abs":18.125,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/norm":7.625,"train/train/layer__model_layers_5/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/norm":0.02011745045314152,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/std":2.7691215876504935e-05,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/mean":-1.341104507446289e-06,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/norm":12.75,"train/train/tensor_act_model_layers_27_self_attn_q_proj/mean":0.019622802734375,"train/train/tensor_act_model_layers_52_self_attn_k_proj/std":0.7812501454353197,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/std":2.9480882152960925e-05,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/std":2.927784226237603e-05,"train/train/tensor_act_model_layers_79_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_v_proj/norm":2025.7895645874312,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/std":0.054443359375,"train/train/tensor_act_model_layers_8_mlp_down_proj/std":0.04010023713042261,"train/train/global/grad/norm":0.732441572585078,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp_down_proj/std":0.15283282528225764,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/max_abs":0.00165557861328125,"train/train/layer_model_layers_47/act/norm":14622.001435018032,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/norm":0.016762125309476555,"train/train/tensor_act_model_layers_13_self_attn_k_proj/std":0.934572263693691,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/max_abs":0.1533203125,"train/train/tensor_act_model_layers_24_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/max_abs":0.00104522705078125,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/std":2.9555459979816207e-05,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/max_abs":0.0003757476806640625,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/mean":1.4246325008571148e-08,"train/train/tensor_act_model_layers_88_self_attn/norm":2257.7352672049547,"train/train/layer_model_layers_5/act/mean":-0.0001773834228515625,"train/train/tensor_act_model_layers_61_mlp_up_proj/mean":-0.095703125,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/norm":0.005570565727545133,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn_q_proj/max_abs":5.53125,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/mean":-9.266659617424011e-08,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn_k_proj/norm":5252.43053735553,"train/train/tensor_act_model_layers_57_self_attn/norm":551.9051980183211,"train/train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/norm":0.01660447936519439,"train/train/tensor_act_model_layers_14_self_attn_k_proj/mean":-0.023193359375,"train/train/tensor_act_model_layers_14_mlp/mean":0.0042266845703125,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/std":0.05078125,"train/train/layer_model_layers_7/act/max_abs":18.5,"train/train/layer_model_layers_25/act/norm":14193.152668810737,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/std":4.045195805505958e-05,"train/train/tensor_act_model_layers_86/norm":11273.474736969345,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/mean":-1.2628734111785889e-06,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/max_abs":0.150390625,"train/train/layer_model_layers_82/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_15/act/max_abs":18.125,"train/train/layer_model_layers_92/act/std":1.087882636251253,"train/train/tensor_act_model_layers_59_self_attn_o_proj/std":0.10380033373155731,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/mean":0.000209808349609375,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_44_self_attn/std":0.09204155333591944,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/max_abs":0.1328125,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71/max_abs":13.25,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/std":5.3678279020433755e-05,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_70/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_43/param/std":0.052823708090701416,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_mlp_up_proj/std":0.287109375,"train/train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn_k_proj/norm":5757.186434569558,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/std":0.02490234375,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/std":0.034423828125,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/std":9.594953489307331e-05,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/std":0.042236328125,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/norm":0.006682879287683888,"train/train/tensor_act_model_layers_2/mean":0.024322509765625,"train/train/tensor_act_model_layers_62_self_attn_o_proj/norm":879.6333991944758,"train/train/layer_model_layers_12/grad/norm":0.024346225955914798,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/mean":1.5832483768463135e-08,"train/train/tensor_act_model_layers_33/frac_near_user_limit":0,"train/train/layer_model_layers_87/act/std":0.8882114058166145,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/std":8.45467696067758e-05,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/mean":-2.905726432800293e-06,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/max_abs":0.2294921875,"train/train/tensor_act_model_layers_74_self_attn/norm":618.9524845050278,"train/train/tensor_act_model_layers_90_self_attn_q_proj/mean":0.05877685546875,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_v_proj/norm":1645.2923150406148,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_83/grad/norm":0.08166384838644661,"train/train/tensor_act_model_layers_59_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_q_proj/mean":-0.029449462890625,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_1/max_abs":18.375,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/std":5.2857709526531955e-05,"train/train/tensor_act_model_layers_72_input_layernorm/mean":0.02349853515625,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp_down_proj/mean":0.0016613006591796875,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_20/grad/std":3.11613524586756e-05,"train/train/tensor_act_model_layers_25_self_attn_q_proj/norm":6253.785347223082,"train/train/tensor_act_model_layers_61_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_input_layernorm/norm":5792.60595703266,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/mean":-4.6156346797943115e-06,"train/train/tensor_act_model_layers_71_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/mean":0.000186920166015625,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/norm":6.65625,"train/train/layer__model_layers_83/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_mlp/max_abs":1.53125,"train/train/tensor_act_model_layers_91_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_52/act/max_abs":15.4375,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/norm":0.006819039986394365,"train/train/tensor_act_model_layers_41_self_attn_o_proj/max_abs":0.5390625,"train/train/layer_model_layers_73/act/norm":14797.25052716305,"train/train/layer_model_layers_81/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/std":0.0439453125,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/mean":-0.0002536773681640625,"train/train/tensor_act_model_layers_17_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_53_self_attn_o_proj/mean":0.0011157989501953125,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/mean":-0.0002574920654296875,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/std":0.0439453125,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/norm":3.265625,"train/train/layer_model_layers_7/grad/max_abs":0.00151824951171875,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53/std":1.3085998122231424,"train/train/tensor_act_model_layers_42_input_layernorm/max_abs":6.125,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_86/act/max_abs":11.8125,"train/train/tensor_act_model_layers_9_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/norm":4.875,"train/train/tensor_param_model_layers_14_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_65/act/std":0.7088675372636047,"train/train/tensor_act_model_layers_42_mlp_down_proj/max_abs":0.7109375,"train/train/layer__model_layers_41/param/std":0.05099807840801883,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/std":0.0223388671875,"train/train/layer_model_layers_87/grad/mean":9.88794330873653e-07,"train/train/tensor_act_model_layers_59_self_attn/norm":600.7061882246599,"train/train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/std":0.037109375,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/mean":0.0002269744873046875,"train/train/tensor_param_model_layers_78_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_33_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn_o_proj/std":0.04339730185057409,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/std":0.048583984375,"train/train/tensor_act_model_layers_35_input_layernorm/norm":5792.612792974317,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/mean":5.4016709327697754e-08,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/norm":0.003367010078712353,"train/train/tensor_act_model_layers_5/std":1.49611927754797,"train/train/tensor_act_model_layers_6_mlp_up_proj/max_abs":2.96875,"train/train/tensor_act_model_layers_83_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_24/act/mean":-0.011994393972250132,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/std":6.78088879995619e-05,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/norm":0.015696065938741816,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/norm":0.0003428309945372701,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/mean":5.832407623529434e-08,"train/train/tensor_act_model_layers_87_self_attn_o_proj/max_abs":3.203125,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/max_abs":0.0001468658447265625,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/max_abs":6.961822509765625e-05,"train/train/tensor_act_model_layers_83_mlp_up_proj/mean":-0.114501953125,"train/train/tensor_act_model_layers_90_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/mean":-0.00010824203491210938,"train/train/tensor_act_model_layers_79_mlp/std":0.1977545167183578,"train/train/layer__model_layers_55/param/mean":0.001377111664056406,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/norm":3.28125,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/max_abs":0.000209808349609375,"train/train/tensor_act_model_layers_39_self_attn/std":0.06207425511598431,"train/train/layer_model_layers_21/act/max_abs":17.875,"train/train/layer__model_layers_21/param/norm":20.018899761381743,"train/train/layer_model_layers_57/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/norm":0.01272864733644374,"train/train/tensor_act_model_layers_23_mlp_down_proj/std":0.04010023718712394,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/max_abs":0.00099945068359375,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/norm":9.25,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/max_abs":0.1806640625,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/max_abs":0.0003070831298828125,"train/train/tensor_act_model_layers_15_self_attn_q_proj/mean":0.042724609375,"train/train/tensor_act_model_layers_82_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/mean":2.728775143623352e-07,"train/train/tensor_act_model_layers_72_mlp_up_proj/std":0.5000005364415152,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_mlp_down_proj/max_abs":1.0234375,"train/train/tensor_act_model_layers_34_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/norm":0.005231272824384604,"train/train/tensor_param_model_layers_15_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_embed_tokens_weight/norm":81.5,"train/train/tensor_act_model_layers_69_self_attn_q_proj/norm":5283.355517421927,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/std":8.732734843943132e-05,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_85_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_down_proj/mean":0.012054443359375,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/mean":-2.130400389432907e-08,"train/train/tensor_act_model_layers_20_mlp_down_proj/norm":217.8237557133209,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/mean":-9.42964106798172e-08,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/mean":-0.0002574920654296875,"train/train/tensor_act_model_layers_28_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/norm":0.0015941099097660973,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_input_layernorm/max_abs":5.875,"train/train/tensor_act_model_layers_91_self_attn_k_proj/mean":0.017547607421875,"train/train/tensor_act_model_layers_34_self_attn_o_proj/std":0.08667096912647912,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/max_abs":0.0002288818359375,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/mean":-3.209710121154785e-05,"train/train/tensor_act_model_layers_93_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn_k_proj/std":0.8369166531072024,"train/train/tensor_act_model_layers_9_self_attn/mean":-0.000759124755859375,"train/train/tensor_act_model_layers_91/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_k_proj/norm":5062.4216559356755,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/norm":5.875,"train/train/tensor_act_model_layers_74_self_attn_o_proj/max_abs":1.1171875,"train/train/tensor_act_model_layers_43_self_attn_q_proj/max_abs":6.84375,"train/train/layer__model_layers_6/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn/norm":719.1592407583274,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_67/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/mean":1.385924406349659e-07,"train/train/tensor_act_model_layers_31_self_attn_o_proj/std":0.05084947545090212,"train/train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn/max_abs":1.0625,"train/train/tensor_act_model_layers_28_input_layernorm/norm":5792.607177735142,"train/train/tensor_act_model_norm/frac_near_user_limit":0,"train/train/layer__model_layers_50/param/std":0.05311695972216611,"train/train/layer_model_layers_43/act/frac_near_user_limit":0,"train/train/layer_model_layers_46/grad/norm":0.043203588984166666,"train/train/tensor_act_model_layers_84_self_attn/max_abs":1.765625,"train/train/tensor_act_model_layers_93_self_attn_q_proj/norm":5449.517005152552,"train/train/tensor_act_model_layers_81_post_attention_layernorm/mean":0.05438232421875,"train/train/tensor_act_model_layers_6_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_post_attention_layernorm/norm":5792.60766601668,"train/train/tensor_act_model_layers_87_post_attention_layernorm/norm":5792.607177738516,"train/train/tensor_act_model_layers_24_self_attn/max_abs":0.71484375,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/norm":0.024456673629722144,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_7_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn/std":0.156739104606625,"train/train/tensor_act_model_layers_25_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/norm":5.625,"train/train/tensor_act_model_layers_81_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/mean":-0.0003795623779296875,"train/train/layer__model_layers_82/param/mean":0.001348581775302262,"train/train/layer_model_layers_86/act/std":0.8618652624385723,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/norm":0.0019910563794347563,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/norm":8.4375,"train/train/tensor_act_model_layers_38/mean":0.0865478515625,"train/train/layer_model_layers_48/grad/std":5.192943027429765e-05,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/max_abs":7.104873657226562e-05,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/max_abs":0.1630859375,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/std":0.0277099609375,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/norm":7.625,"train/train/layer_model_layers_33/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/std":0.0289306640625,"train/train/tensor_act_model_layers_41_self_attn/max_abs":0.5390625,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_40/grad/mean":-6.541832787523776e-07,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/norm":0.008432465746855529,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/mean":-0.0002536773681640625,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/norm":5.75,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61/max_abs":13.6875,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_input_layernorm/mean":0.0655517578125,"train/train/tensor_act_model_layers_14_self_attn_o_proj/max_abs":0.6953125,"train/train/layer_model_layers_63/grad/max_abs":0.001434326171875,"train/train/tensor_act_model_layers_12_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_39/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/std":0.00013877735954747857,"train/train/tensor_act_model_layers_45_post_attention_layernorm/max_abs":5.78125,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/max_abs":0.000705718994140625,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/norm":0.012728574168790748,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/norm":4.6875,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/mean":-0.00022411346435546875,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_52/param/std":0.05219820677151281,"train/train/tensor_act_model_layers_19_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/mean":0.0002346038818359375,"train/train/tensor_act_model_layers_19_post_attention_layernorm/mean":0.022979736328125,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/max_abs":0.00055694580078125,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/max_abs":0.19921875,"train/train/tensor_act_model_layers_8_self_attn/norm":406.01484164935357,"train/train/tensor_act_model_layers_37_mlp/norm":341.4315058516043,"train/train/tensor_act_model_layers_49_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/norm":0.015397532070177302,"train/train/tensor_act_model_layers_4_self_attn/std":0.049256506181702756,"train/train/layer_model_layers_4/grad/mean":-3.9479235626590786e-07,"train/train/tensor_act_model_layers_45_self_attn/mean":-0.0002086162567138672,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/norm":7.0625,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/max_abs":0.000972747802734375,"train/train/tensor_act_model_layers_30_mlp/norm":264.15771479480543,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/std":0.02490234375,"train/train/layer_model_layers_20/grad/max_abs":0.00127410888671875,"train/train/layer_model_layers_37/act/max_abs":17.125,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/mean":1.318258000537753e-07,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/max_abs":0.000843048095703125,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/mean":-0.0002689361572265625,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/std":3.7239513747788086e-05,"train/train/tensor_act_model_layers_22_mlp/max_abs":0.5390625,"train/train/tensor_act_model_layers_4_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp/max_abs":2.234375,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/std":2.214991953174912e-05,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/std":5.6030078111018e-05,"train/train/tensor_param_model_layers_45_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/norm":5.125,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/std":2.59237400994079e-05,"train/train/tensor_act_model_layers_3_mlp_up_proj/std":0.27734383059218404,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/mean":3.7834979593753815e-08,"train/train/tensor_act_model_layers_57_input_layernorm/norm":5792.602783206258,"train/train/tensor_act_model_layers_53/mean":0.05877685546875,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/max_abs":0.00010776519775390625,"train/train/layer__model_layers_83/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_76_mlp_down_proj/max_abs":1.78125,"train/train/tensor_act_model_layers_0_input_layernorm/mean":0.0191650390625,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/max_abs":0.1279296875,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/mean":-0.0001697540283203125,"train/train/tensor_act_model_layers_28_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/norm":6.46875,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/std":0.03515625,"train/train/tensor_param_model_layers_62_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/std":0.00014496902210752415,"train/train/tensor_act_model_layers_60/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/mean":0.03057861328125,"train/train/tensor_act_model_layers_89_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_q_proj/norm":5840.041521492557,"train/train/tensor_act_model_layers_11/std":1.4316659835176613,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/std":0,"train/train/layer__model_layers_87/param/mean":0.0012810717506825274,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/std":3.563770818091314e-05,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/mean":4.961912054568529e-07,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33/mean":0.073974609375,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/mean":3.889435902237892e-07,"train/train/tensor_act_model_layers_79/norm":9795.179215655104,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/norm":0.028767961198964118,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/norm":0.014651679633885563,"train/train/tensor_act_model_layers_74_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90/norm":13411.896925509032,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/std":0.0517578125,"train/train/layer_model_layers_84/act/max_abs":11.5,"train/train/tensor_param_model_layers_24_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/std":0.02294921875,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_q_proj/max_abs":5.875,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn/std":0.09510203024815737,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/norm":0.014705134974975376,"train/train/layer_model_layers_73/grad/max_abs":0.00177001953125,"train/train/tensor_act_model_layers_89_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/max_abs":0.000579833984375,"train/train/layer_model_layers_43/act/std":0.6789855908041771,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/norm":0.01236425242293207,"train/train/tensor_act_model_layers_20_self_attn_k_proj/max_abs":4.90625,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/max_abs":0.000553131103515625,"train/train/layer__model_layers_23/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/max_abs":0.0004863739013671875,"train/train/tensor_act_model_layers_42_mlp_up_proj/mean":-0.08056640625,"train/train/tensor_act_model_layers_47_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_22/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_76/act/std":0.7450864926301863,"train/train/tensor_act_model_layers_24_self_attn_q_proj/mean":-0.0411376953125,"train/train/tensor_act_model_layers_52_self_attn_v_proj/norm":1925.5740660179051,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/max_abs":0.12060546875,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/std":0.0341796875,"train/train/layer__model_layers_61/param/norm":22.29439479022698,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/std":0.0341796875,"train/train/tensor_act_model_layers_69/norm":8470.16995409056,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/norm":6.53125,"train/train/tensor_act_model_layers_0_self_attn_v_proj/mean":-0.000743865966796875,"train/train/layer__model_layers_76/param/mean":0.0013234916603695399,"train/train/tensor_act_model_layers_7_post_attention_layernorm/mean":-0.00510406494140625,"train/train/tensor_act_model_layers_15_self_attn/max_abs":1.2890625,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/std":0.05029296875,"train/train/tensor_act_model_layers_14_post_attention_layernorm/mean":0.004093170166015625,"train/train/tensor_act_model_layers_76_mlp/max_abs":1.78125,"train/train/tensor_act_model_layers_87_self_attn_v_proj/norm":2813.768485607098,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/std":0.0283203125,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_41_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/std":4.7507289450992333e-05,"train/train/layer_model_layers_72/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/norm":6.53125,"train/train/tensor_param_model_layers_58_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/std":4.9530007574755075e-05,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/std":1.3152520603017288e-05,"train/train/tensor_act_model_layers_65_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_q_proj/max_abs":6.78125,"train/train/tensor_act_model_layers_16_input_layernorm/mean":0.0134124755859375,"train/train/tensor_act_model_layers_32_input_layernorm/max_abs":6.125,"train/train/tensor_act_model_layers_92_input_layernorm/norm":5792.610595707091,"train/train/tensor_act_model_layers_1_self_attn_o_proj/norm":541.5968840862648,"train/train/tensor_act_model_layers_14_self_attn_q_proj/std":1.0156254183786704,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/mean":0.000858306884765625,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/max_abs":0.00089263916015625,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/std":0.046630859375,"train/train/layer__model_layers_13/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp_up_proj/std":0.2080078125,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/mean":-0.000946044921875,"train/train/tensor_param_model_layers_31_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_mlp_down_proj/std":0.18310620879977402,"train/train/tensor_act_model_layers_50_self_attn_q_proj/norm":5253.076538281274,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_mlp_down_proj/std":0.04632613765483977,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/norm":0.02794457152389145,"train/train/layer__model_layers_12/param/mean":0.0016569235767477574,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/norm":3.265625,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/norm":6.90625,"train/train/layer__model_layers_59/param/max_abs":1,"train/train/tensor_act_model_layers_71_self_attn/std":0.07840758077749085,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/max_abs":0.205078125,"train/train/tensor_act_model_layers_42_mlp_down_proj/norm":362.6507540354738,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/mean":8.230745152104646e-08,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/mean":1.4819670468568802e-07,"train/train/tensor_act_model_layers_37_self_attn_v_proj/mean":0.003360748291015625,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/mean":-0.00040435791015625,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/max_abs":0.000522613525390625,"train/train/tensor_param_model_layers_29_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_post_attention_layernorm/max_abs":6.15625,"train/train/layer_model_layers_21/act/std":0.6849715946046192,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_40_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/mean":1.0493749869056046e-08,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/std":6.858796835609655e-06,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_k_proj/norm":5249.510868317152,"train/train/tensor_act_model_layers_29_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17/std":1.3730622245939885,"train/train/tensor_act_model_layers_33_mlp_up_proj/mean":-0.072265625,"train/train/tensor_param_model_layers_43_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_82/act/max_abs":11.875,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/std":6.592962269932557e-05,"train/train/tensor_act_model_layers_1_input_layernorm/std":1.0000000223517413,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_43_self_attn_o_proj/max_abs":1.703125,"train/train/tensor_param_model_layers_92_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_q_proj/max_abs":5.5,"train/train/tensor_act_model_layers_38_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/norm":6.6875,"train/train/layer_model_layers_84/grad/std":9.905141228679755e-05,"train/train/tensor_act_model_layers_84_self_attn_o_proj/mean":-0.002216339111328125,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/std":2.145352552633094e-05,"train/train/tensor_act_model_layers_90_mlp_up_proj/mean":-0.158935546875,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/std":0.024169921875,"train/train/tensor_act_model_layers_39_post_attention_layernorm/max_abs":6.0625,"train/train/layer__model_layers_3/param/mean":0.001267265046070593,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/norm":0.0027362310547530055,"train/train/layer_model_layers_79/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/norm":3.8125,"train/train/tensor_act_model_layers_88/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_input_layernorm/norm":5792.604614263465,"train/train/tensor_act_model_layers_46/mean":0.0830078125,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/std":3.178539147267024e-05,"train/train/layer_model_layers_30/act/norm":14487.967968382392,"train/train/tensor_act_model_layers_28_self_attn_v_proj/norm":1772.576182001147,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/std":4.238708022502205e-05,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/norm":5.84375,"train/train/tensor_act_model_layers_50_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/norm":0.025154630952469764,"train/train/layer__model_layers_61/param/std":0.055063092294910014,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/max_abs":0.0003509521484375,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/max_abs":0.00015354156494140625,"train/train/tensor_act_model_layers_23_mlp/max_abs":0.72265625,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/max_abs":0.2353515625,"train/train/tensor_act_model_layers_54_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_22/act/norm":14383.493303862946,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_48/grad/mean":-6.748395975586172e-07,"train/train/layer__model_layers_34/param/std":0.05073909200841603,"train/train/tensor_act_model_layers_57_mlp_down_proj/std":0.09887726218563803,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/mean":-2.1550804376602173e-06,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/mean":-0.00063323974609375,"train/train/tensor_act_model_layers_83_self_attn_k_proj/max_abs":6.28125,"train/train/tensor_param_model_layers_90_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/norm":0.02144879950296183,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/mean":-0.00086212158203125,"train/train/tensor_act_model_layers_48_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_q_proj/norm":6337.051286075502,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/norm":5.5,"train/train/tensor_act_model_layers_18_mlp_up_proj/std":0.21972669813363868,"train/train/layer_model_layers_19/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/max_abs":0.000904083251953125,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/std":3.117842684467793e-05,"train/train/layer_model_layers_76/act/mean":-0.015248518723707933,"train/train/tensor_act_model_layers_21_self_attn_o_proj/max_abs":0.671875,"train/train/tensor_act_model_layers_18_post_attention_layernorm/max_abs":6.125,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/max_abs":0.0002918243408203125,"train/train/tensor_act_model_layers_30_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/max_abs":5,"train/train/layer_model_layers_42/act/std":0.6694717650130507,"train/train/tensor_act_model_layers_66_mlp_down_proj/max_abs":1.1796875,"train/train/tensor_act_model_layers_50_self_attn_q_proj/mean":0.04376220703125,"train/train/tensor_act_model_layers_6/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/mean":-0.0001430511474609375,"train/train/layer_model_layers_74/grad/mean":1.6230758564334577e-06,"train/train/layer__model_layers_25/param/mean":0.0016329998903081124,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/std":3.6320904120238005e-05,"train/train/layer__model_layers_54/param/mean":0.0015002650887285492,"train/train/layer_model_layers_5/act/std":0.786995644363864,"train/train/tensor_act_model_layers_47_input_layernorm/norm":5792.6093750007,"train/train/layer_model_layers_27/act/std":0.6657023143529632,"train/train/layer_model_layers_85/act/std":0.8261599114962337,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/std":2.97751085331167e-05,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_54/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/mean":-0.0001888275146484375,"train/train/tensor_act_/std":0,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/max_abs":0.0002803802490234375,"train/train/tensor_param_model_layers_43_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_63_self_attn_k_proj/mean":-0.048095703125,"train/train/tensor_act_model_layers_79_mlp_up_proj/mean":-0.1014404296875,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/mean":3.935769200325012e-06,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/mean":-3.23285348713398e-07,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/mean":0.00022983551025390625,"train/train/tensor_act_model_layers_52_input_layernorm/norm":5792.612182621668,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/mean":1.685693860054016e-07,"train/train/tensor_act_model_layers_51_self_attn_v_proj/max_abs":2.421875,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/mean":-2.1043233573436737e-06,"train/train/tensor_act_model_layers_56_mlp_down_proj/norm":486.27301085125583,"train/train/layer__model_layers_84/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_15/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn_q_proj/max_abs":5.8125,"train/train/tensor_param_model_layers_30_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_15/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/std":3.3270574955146046e-05,"train/train/tensor_param_model_layers_3_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/norm":0.010897040465586183,"train/train/tensor_act_model_layers_40_post_attention_layernorm/max_abs":5.96875,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/mean":-4.253524821251631e-08,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/max_abs":7.295608520507812e-05,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/std":0.044921875,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_4/grad/norm":0.04050867342234139,"train/train/tensor_act_model_layers_59_mlp_down_proj/std":0.1077883408263704,"train/train/tensor_param_model_layers_26_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_45/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp/std":0.1086428829286618,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/max_abs":0.000583648681640625,"train/train/layer_model_layers_55/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/mean":-0.0004825592041015625,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/mean":-2.075103111565113e-08,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/std":3.060095822054074e-05,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/std":0.025390625,"train/train/tensor_param_model_layers_3_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/std":4.8940587921877024e-05,"train/train/layer__model_layers_85/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/max_abs":0.10595703125,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/max_abs":0.07763671875,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/norm":0.010624763962984878,"train/train/tensor_act_model_layers_21_self_attn_v_proj/mean":0.00527191162109375,"train/train/tensor_act_model_layers_89_self_attn_q_proj/std":1.080084112609491,"train/train/layer__model_layers_12/param/max_abs":1,"train/train/tensor_act_model_layers_93_self_attn_q_proj/mean":0.07177734375,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/std":5.051446374377996e-05,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/mean":-7.101334631443024e-09,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/std":6.676328863543826e-05,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_29/act/mean":0.004018824834090013,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/mean":-2.33575701713562e-06,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/std":0.03076171875,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/norm":0.005182487713318009,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/mean":-6.580352783203125e-05,"train/train/tensor_act_model_layers_37/mean":0.0987548828125,"train/train/tensor_act_model_layers_43_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_41/grad/mean":-7.37085880336821e-07,"train/train/tensor_act_model_layers_83_self_attn_o_proj/norm":1067.5138976407538,"train/train/tensor_act_model_layers_45_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_v_proj/mean":0.002819061279296875,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/max_abs":0.1337890625,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/max_abs":0.001007080078125,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/max_abs":0.00026702880859375,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/max_abs":0.002655029296875,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/mean":-0.00034332275390625,"train/train/tensor_act_model_layers_4/mean":0.0053558349609375,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/max_abs":0.0002574920654296875,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/mean":-1.6596168279647827e-06,"train/train/tensor_param_model_layers_14_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/mean":-4.437752068042755e-07,"train/train/tensor_act_model_layers_11_mlp_down_proj/max_abs":0.310546875,"train/train/tensor_act_model_layers_5_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/max_abs":0.00125885009765625,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/norm":4.875,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/mean":-6.742775440216064e-07,"train/train/tensor_act_model_layers_92_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_90/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/norm":4.90625,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/norm":5.0625,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/mean":-9.696930646896362e-06,"train/train/tensor_act_model_layers_72_self_attn_o_proj/mean":0.002948760986328125,"train/train/tensor_param_model_layers_51_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/mean":0.00063323974609375,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/norm":6.9375,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/mean":2.50060111284256e-06,"train/train/tensor_act_model_layers_79_mlp_up_proj/max_abs":3.390625,"train/train/layer_model_layers_16/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/max_abs":0.00030517578125,"train/train/tensor_act_model_layers_24_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_92/grad/mean":-3.251129421354642e-07,"train/train/tensor_act_model_layers_60_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_25/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_input_layernorm/max_abs":5.875,"train/train/tensor_param_model_layers_50_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/mean":-1.6521662473678589e-06,"train/train/tensor_act_model_layers_21_mlp/max_abs":0.7109375,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/mean":-0.000904083251953125,"train/train/layer_model_layers_75/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_49/act/mean":-0.013681265024038462,"train/train/tensor_act_model_layers_41_input_layernorm/norm":5792.603271487323,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/max_abs":0.0002460479736328125,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/std":0.0281982421875,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/max_abs":0.0029144287109375,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/mean":4.4889748096466064e-06,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_76/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57/mean":0.05126953125,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_post_attention_layernorm/std":1.0000000190921126,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_q_proj/mean":-0.03948974609375,"train/train/tensor_param_model_layers_61_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/std":0.052734375,"train/train/tensor_act_model_layers_29_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_49/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/norm":4.09375,"train/train/tensor_act_model_layers_93_self_attn_q_proj/max_abs":6.21875,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/std":6.502434446710455e-05,"train/train/layer_model_layers_21/act/norm":14304.965025074629,"train/train/layer_model_layers_40/act/max_abs":16.875,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/norm":0.013454791465360261,"train/train/layer_model_layers_54/grad/max_abs":0.00069427490234375,"train/train/tensor_act_model_layers_45_self_attn_v_proj/max_abs":2.296875,"train/train/layer__model_layers_35/param/norm":20.594046374953855,"train/train/tensor_act_model_layers_46_self_attn_k_proj/std":0.8906250920212012,"train/train/tensor_act_model_layers_89_self_attn_o_proj/mean":-0.01202392578125,"train/train/tensor_act_model_layers_66_self_attn/norm":1192.525413421562,"train/train/tensor_act_model_layers_73_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/max_abs":0.2294921875,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/mean":-2.1904706954956055e-06,"train/train/layer__model_layers_69/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11/norm":8290.749024129498,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_76_mlp_down_proj/std":0.16650463316070407,"train/train/tensor_act_model_layers_31/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/mean":0.0003509521484375,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/mean":0.00052642822265625,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/max_abs":0.000736236572265625,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/norm":0.005922336854764868,"train/train/layer__model_layers_64/param/max_abs":1,"train/train/tensor_param_model_layers_85_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_6_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_input_layernorm/norm":5792.6110839899575,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/mean":-3.2084062695503235e-07,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/norm":0.024483087239241558,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/max_abs":0.00127410888671875,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/norm":0.002957982280743175,"train/train/tensor_act_model_layers_71_self_attn_v_proj/norm":2108.7680786390215,"train/train/tensor_act_model_layers_86_mlp_down_proj/max_abs":3.203125,"train/train/tensor_act_model_layers_59_self_attn_v_proj/norm":2139.1682739531843,"train/train/tensor_act_model_layers_25_self_attn_q_proj/mean":-0.0750732421875,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/mean":0.0003032684326171875,"train/train/tensor_act_model_layers_92_self_attn_k_proj/mean":0.0625,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_k_proj/norm":7206.072309005241,"train/train/tensor_act_model_layers_54_mlp_up_proj/norm":4045.144779113782,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_90/grad/norm":0.09594909515165374,"train/train/tensor_act_lm_head/std":2.097681709575478,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/mean":1.41095370054245e-07,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/mean":-0.0005645751953125,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/std":1.93463528070714e-05,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/norm":0.008971441330322507,"train/train/tensor_act_model_layers_74_self_attn_o_proj/std":0.10681408754319424,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/std":0.031982421875,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_22/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/max_abs":0.0002231597900390625,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/max_abs":0.0004177093505859375,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn/mean":-0.0006551742553710938,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/mean":1.1210795491933823e-05,"train/train/tensor_act_model_layers_28_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn_o_proj/norm":1044.2974707242477,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/max_abs":0.0003185272216796875,"train/train/tensor_act_model_layers_69_self_attn/mean":-0.000797271728515625,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/mean":-3.248453140258789e-06,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/norm":0.004091916776734881,"train/train/tensor_act_model_layers_32_mlp_up_proj/max_abs":2.03125,"train/train/tensor_act_model_layers_38_self_attn_k_proj/std":0.9726603816224251,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/max_abs":0.0002727508544921875,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/norm":7.375,"train/train/layer__model_layers_16/param/norm":19.36194746831075,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/std":1.4763206748729355e-05,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/max_abs":0.00015544891357421875,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/std":0.053955078125,"train/train/tensor_act_model_layers_59_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_mlp_up_proj/mean":-0.0830078125,"train/train/tensor_act_model_layers_21_self_attn_q_proj/max_abs":5.1875,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn_k_proj/norm":5283.715487519011,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/std":0.055419921875,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/norm":0.030488328267757075,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/mean":-1.3899989426136017e-06,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/max_abs":0.000789642333984375,"train/train/tensor_act_model_layers_31/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn/max_abs":1.2421875,"train/train/tensor_act_model_layers_58_self_attn_v_proj/mean":-0.00032019615173339844,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_69/grad/mean":-1.9816757294577482e-07,"train/train/tensor_act_model_layers_15_self_attn_q_proj/norm":6119.269121999995,"train/train/tensor_act_model_layers_6_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/mean":-4.7497451305389404e-06,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/max_abs":0.000522613525390625,"train/train/tensor_act_model_layers_29_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/norm":0.004538046984241654,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/mean":9.19681042432785e-09,"train/train/tensor_act_model_layers_83_mlp/norm":1344.3798457180258,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_29_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_13/param/frac_near_user_limit":0,"train/train/layer__model_layers_70/param/norm":25.220260177831236,"train/train/layer_model_layers_79/act/std":0.791086447215829,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/mean":0.000194549560546875,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/std":4.5788680275426433e-05,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/std":0.055419921875,"train/train/layer_model_layers_73/grad/mean":1.5770625642699867e-06,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/mean":-1.8775463104248047e-06,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/mean":-2.1651387214660645e-05,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/mean":1.7952173948287964e-05,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/std":0.051513671875,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/norm":5.6875,"train/train/tensor_param_model_layers_49_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_norm_weight/norm":11.3125,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/std":0.049072265625,"train/train/tensor_act_model_layers_33_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_input_layernorm/norm":5792.608398443489,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/max_abs":0.007568359375,"train/train/tensor_act_model_layers_50_self_attn_o_proj/max_abs":0.8515625,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/std":0.047607421875,"train/train/tensor_act_model_layers_26_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp/std":0.04010023713042261,"train/train/tensor_act_model_layers_47_self_attn_k_proj/max_abs":4.90625,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/max_abs":0.000942230224609375,"train/train/tensor_act_model_layers_69_self_attn_q_proj/mean":0.00691986083984375,"train/train/tensor_act_model_layers_30_self_attn_o_proj/std":0.10974200919924976,"train/train/tensor_act_model_layers_70_input_layernorm/mean":0.029022216796875,"train/train/tensor_act_model_layers_30_mlp/std":0.04541049144477597,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/std":9.759142487727226e-05,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/max_abs":0.00019359588623046875,"train/train/tensor_act_model_layers_79_self_attn_v_proj/norm":2695.6438762624807,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/mean":-2.4829059839248657e-06,"train/train/tensor_act_model_layers_9_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/std":0.048828125,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/mean":1.1431984603404999e-07,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/norm":0.04079669469927214,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/max_abs":0.00052642822265625,"train/train/tensor_act_model_layers_71_input_layernorm/max_abs":5.9375,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/norm":0.0004970143946356445,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/std":0.00010897741182450703,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_post_attention_layernorm/norm":5792.608032226968,"train/train/tensor_act_model_layers_38_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8/norm":8459.911511048447,"train/train/tensor_act_model_layers_91/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_57/grad/norm":0.04946077654781701,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_o_proj/mean":-0.0002884864807128906,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/max_abs":0.220703125,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/max_abs":0.10791015625,"train/train/tensor_act_model_layers_76_input_layernorm/std":1.0000027622989411,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/std":2.7891155419246057e-05,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_56/param/max_abs":1,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/max_abs":0.00107574462890625,"train/train/layer_model_layers_23/act/mean":-0.013400444617638221,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp_up_proj/max_abs":3.046875,"train/train/tensor_grad_model_norm_weight/mean":-0.00296783447265625,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_q_proj/norm":5609.7128371864455,"train/train/tensor_act_model_layers_49_mlp_down_proj/max_abs":0.90234375,"train/train/tensor_act_model_layers_21_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/max_abs":0.0006256103515625,"train/train/tensor_param_model_layers_77_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/norm":0.001889218951858555,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/max_abs":0.2490234375,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/max_abs":0.1572265625,"train/train/layer_model_layers_15/act/mean":-0.008428426889272837,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/mean":-1.9849976524710655e-06,"train/train/tensor_act_model_layers_44_mlp_up_proj/max_abs":3.28125,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/std":1.237605921819189e-05,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/norm":0.0050739775740563375,"train/train/tensor_act_model_layers_11_self_attn_v_proj/norm":1555.799142035676,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/norm":0.013707115531829685,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/mean":-1.6426201909780502e-07,"train/train/tensor_act_model_layers_14_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp_up_proj/mean":-0.0977783203125,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_18/act/mean":0.003357538810143104,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/mean":-2.176966518163681e-08,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/mean":7.88422767072916e-08,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_47/grad/norm":0.053562992185453254,"train/train/tensor_param_model_layers_15_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_76_self_attn_o_proj/norm":728.9427564260864,"train/train/tensor_act_model_layers_67_self_attn_k_proj/std":0.8632813060984874,"train/train/layer__model_layers_52/param/norm":21.153630284628452,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/std":5.2406202113659135e-05,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/std":3.212935840611045e-05,"train/train/tensor_act_model_layers_54_self_attn_v_proj/mean":-0.002899169921875,"train/train/tensor_act_model_layers_64_self_attn_o_proj/norm":909.4365599799609,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/norm":4.78125,"train/train/tensor_act_model_layers_42_mlp_up_proj/std":0.3281255449562997,"train/train/tensor_act_model_layers_14_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/std":5.975534853700488e-05,"train/train/tensor_act_model_layers_10_mlp_up_proj/norm":2246.8092367554414,"train/train/tensor_act_model_layers_23_self_attn_o_proj/mean":-0.0007476806640625,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/max_abs":0.00096893310546875,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/mean":8.230563253164291e-06,"train/train/tensor_act_model_layers_47_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/max_abs":0.0009613037109375,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/max_abs":0.000499725341796875,"train/train/layer_model_layers_60/act/max_abs":14,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/max_abs":9.1552734375e-05,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/max_abs":0.00021648406982421875,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/norm":0.023386741351449143,"train/train/layer__model_layers_16/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/norm":8.5,"train/train/layer_model_layers_3/act/std":0.8538823920582396,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/std":6.687276573367274e-05,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/norm":6.96875,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/std":0.036865234375,"train/train/tensor_act_model_layers_32_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/max_abs":0.00135040283203125,"train/train/tensor_act_model_layers_42_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/mean":-0.00020599365234375,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/std":5.744331066434625e-05,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_38/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn_o_proj/mean":0.0006642341613769531,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/max_abs":0.000606536865234375,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_q_proj/norm":5476.70520677306,"train/train/tensor_act_model_layers_92_mlp_down_proj/mean":0.0206298828125,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_input_layernorm/max_abs":5.5625,"train/train/tensor_act_model_layers_72/max_abs":13.25,"train/train/tensor_act_model_layers_78_self_attn_q_proj/norm":5650.92250799765,"train/train/layer__model_layers_41/param/max_abs":1,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_64/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/mean":-1.6596168279647827e-06,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/mean":-3.6065466701984406e-07,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/std":4.4900562665987174e-05,"train/train/tensor_act_model_layers_26_mlp_down_proj/mean":0.0111846923828125,"train/train/tensor_param_model_layers_23_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_50_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_o_proj/max_abs":2.71875,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_k_proj/max_abs":5.0625,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_v_proj/std":0.30468750561181546,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/mean":-0.00040435791015625,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39/std":1.3027509605347074,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/std":0.0252685546875,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/norm":0.003075601888256604,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/norm":0.0008913034242758244,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/norm":0.016763569836537637,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_31/grad/norm":0.03319284593558431,"train/train/tensor_act_model_layers_31_self_attn_v_proj/max_abs":1.8515625,"train/train/tensor_act_model_layers_79/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_post_attention_layernorm/mean":-0.0079345703125,"train/train/tensor_act_model_layers_69_self_attn_k_proj/std":0.8808616814720593,"train/train/tensor_act_model_layers_58_self_attn_v_proj/max_abs":3.359375,"train/train/tensor_act_model_layers_37_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/mean":8.01868736743927e-07,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/max_abs":0.2421875,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/std":1.586791519732152e-05,"train/train/tensor_act_model_layers_19_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_v_proj/max_abs":1.984375,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/norm":0.0011971708862321415,"train/train/tensor_act_model_layers_4_self_attn_v_proj/mean":-0.0095367431640625,"train/train/tensor_act_model_layers_3_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/norm":0.021462776424691844,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/norm":5.75,"train/train/tensor_act_model_layers_77_self_attn_v_proj/mean":-0.00022661685943603516,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/norm":0.023471092945285897,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/norm":0.006523805996200795,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/norm":0.007150415735687786,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/std":4.019644659089492e-05,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/std":0.00018334850791258537,"train/learning_rate":0.001,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_15_mlp_up_proj/norm":2358.115689602616,"train/train/tensor_act_model_layers_44_mlp/std":0.06982423708988829,"train/train/tensor_act_model_layers_60_self_attn_v_proj/max_abs":2.734375,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/std":0.03759765625,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/std":0.0283203125,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/std":0.030029296875,"train/train/layer__model_layers_5/param/norm":19.610314967511357,"train/train/tensor_act_model_layers_12_input_layernorm/mean":0.0001201629638671875,"train/train/tensor_act_model_layers_61_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp/norm":444.3641786070455,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/mean":0.00018024444580078125,"train/train/tensor_act_model_layers_71_mlp_up_proj/norm":5110.704702973729,"train/train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer__model_layers_17/param/norm":19.64120300462143,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/mean":-1.7881393432617188e-06,"train/train/tensor_act_model_layers_0_mlp_down_proj/norm":8421.160495516604,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/std":4.83979544899196e-05,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/norm":5.5625,"train/train/tensor_act_model_layers_66_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_mlp_down_proj/mean":0.002197265625,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn_q_proj/std":1.0625001875793068,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/norm":0.01699837319920943,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_63/mean":0.072021484375,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/norm":0.1798108548224019,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/max_abs":0.00012683868408203125,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/norm":3.71875,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/norm":0.003768480715355575,"train/train/layer__model_layers_29/param/max_abs":1,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_post_attention_layernorm/norm":5792.614257814516,"train/train/tensor_param_model_layers_56_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_42_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/mean":-0.0003757476806640625,"train/train/tensor_act_model_layers_32_self_attn/norm":238.1257958749373,"train/train/tensor_param_model_layers_65_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/norm":0.03129746884921965,"train/train/tensor_act_model_layers_74_self_attn_k_proj/mean":0.042236328125,"train/train/tensor_act_model_layers_78_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn/max_abs":1.703125,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/mean":8.881092071533203e-06,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/mean":0.0010528564453125,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp_up_proj/norm":2282.4499836222917,"train/train/tensor_act_model_layers_53_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/std":5.619628616970913e-05,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn_k_proj/max_abs":5,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/std":0.046142578125,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/std":1.685275052467733e-05,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/std":0.00011805192023655671,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/max_abs":0.0003910064697265625,"train/train/tensor_act_model_layers_72/frac_near_user_limit":0,"train/train/layer__model_layers_56/param/mean":0.0012957830324931748,"train/train/tensor_act_model_layers_39_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/std":5.913837274631789e-05,"train/train/tensor_act_model_layers_19_self_attn/mean":-0.0003561973571777344,"train/train/tensor_act_model_layers_90_self_attn_o_proj/std":0.1552814558243816,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/max_abs":0.00095367431640625,"train/train/tensor_act_model_layers_7_self_attn_o_proj/max_abs":0.953125,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_36/max_abs":17.375,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/max_abs":0.150390625,"train/train/tensor_act_model_layers_9_mlp/norm":269.68675804093186,"train/train/tensor_act_model_layers_24/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/max_abs":0.2294921875,"train/train/layer_model_layers_23/act/max_abs":17.75,"train/train/layer__model_layers_10/param/norm":19.756303384932618,"train/train/tensor_act_model_layers_50_self_attn_k_proj/std":0.7812501162290487,"train/train/layer_model_layers_87/act/max_abs":12.125,"train/train/tensor_act_model_layers_21_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp/max_abs":1.6875,"train/train/tensor_act_model_layers_41_self_attn_k_proj/norm":4870.926088606874,"train/train/tensor_act_model_layers_20_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/norm":0.022494167128137676,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_input_layernorm/norm":5792.605346680504,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/mean":-0.0003604888916015625,"train/train/tensor_act_model_layers_12_self_attn_k_proj/std":0.7978535865748678,"train/train/tensor_act_model_layers_0_mlp/std":1.4531565478192037,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/std":0.041259765625,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/max_abs":0.1455078125,"train/train/tensor_act_model_layers_89_mlp_down_proj/std":0.44531688353405463,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/std":0.060791015625,"train/train/tensor_act_model_layers_2/std":1.5078428773710664,"train/train/tensor_param_model_layers_19_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_93/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/mean":-1.7505954019725323e-08,"train/train/tensor_act_model_layers_52_self_attn/max_abs":1.0625,"train/train/layer_model_layers_61/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp/norm":1303.9379899180044,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_q_proj/std":0.9443374917807505,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/max_abs":0.255859375,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/std":0.0537109375,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/std":8.413348456217656e-05,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/std":4.3172636463632055e-05,"train/train/layer_model_layers_6/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_up_proj/std":0.2539068075320651,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/mean":-0.00013828277587890625,"train/train/tensor_act_model_layers_40_self_attn_k_proj/norm":4880.550001869899,"train/train/tensor_act_model_layers_14_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_80/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_up_proj/max_abs":3.03125,"train/train/tensor_act_model_layers_82_mlp/frac_near_user_limit":0,"train/train/layer__model_layers_68/param/norm":23.005136994267172,"train/train/layer__model_layers_93/param/mean":0.001694411457793761,"train/train/layer_model_layers_81/grad/norm":0.07924676760870837,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/mean":0.00040435791015625,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/mean":-3.1152740120887756e-07,"train/train/tensor_act_model_layers_30_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_mlp_down_proj/max_abs":0.8359375,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/norm":0.013905404043719516,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/norm":0.0006343905780625377,"train/train/tensor_act_model_layers_60_input_layernorm/max_abs":5.90625,"train/train/tensor_act_model_layers_74_mlp_down_proj/std":0.1596702173057205,"train/train/layer__model_layers_45/param/max_abs":1,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_67_mlp/norm":790.9359569796349,"train/train/tensor_act_model_layers_39_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_v_proj/norm":2456.8960651821317,"train/train/tensor_act_model_layers_83_self_attn_q_proj/norm":6524.899573037383,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/max_abs":0.1376953125,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_58/act/mean":-0.014302748900193434,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/mean":4.607439041137695e-05,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/std":0.0001425170996885988,"train/train/tensor_act_model_layers_89_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/mean":9.9127646535635e-08,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/mean":-0.000125885009765625,"train/train/tensor_act_model_layers_14_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/mean":9.73668647930026e-08,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_v_proj/mean":-0.027496337890625,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/norm":5.90625,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/std":0.0306396484375,"train/train/tensor_act_model_layers_81_self_attn_k_proj/mean":0.0157012939453125,"train/train/tensor_param_model_layers_30_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/std":7.169296498936297e-05,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/norm":5.25,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/mean":2.2351741790771484e-08,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/std":0.0244140625,"train/train/layer_model_layers_75/grad/norm":0.07127036858351252,"train/train/tensor_act_model_layers_42_self_attn_v_proj/std":0.34179699855189133,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/mean":-0.000728607177734375,"train/train/tensor_act_model_layers_85_mlp_down_proj/mean":0.016754150390625,"train/train/layer_model_layers_92/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/std":1.7775676945099316e-05,"train/train/layer__model_layers_18/param/mean":0.001769858849773913,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/std":0.05322265625,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_66_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3/norm":8738.545675264897,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/std":0.0478515625,"train/train/tensor_act_model_layers_19_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_input_layernorm/std":1.0000020563581307,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_k_proj/norm":4866.836382479708,"train/train/tensor_act_model_layers_84_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer__model_layers_70/param/mean":0.0013739456438609107,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_44/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/mean":0.0001125335693359375,"train/train/tensor_act_model_layers_92_post_attention_layernorm/max_abs":5.09375,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp_down_proj/mean":0.0037994384765625,"train/train/tensor_act_model_layers_82_self_attn/norm":643.2306815338605,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/mean":-2.3283064365386963e-05,"train/train/tensor_act_model_layers_78_input_layernorm/norm":5792.607910157588,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/max_abs":0.21875,"train/train/layer_model_layers_68/grad/max_abs":0.002838134765625,"train/train/layer_model_layers_81/act/mean":-0.00762024292579064,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_post_attention_layernorm/norm":5792.607055666814,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/mean":-0.0004177093505859375,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/norm":6.46875,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/std":0.0291748046875,"train/train/tensor_param_model_layers_84_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_input_layernorm/max_abs":5.6875,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_41/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_88/act/std":0.9535765165706224,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/max_abs":0.000335693359375,"train/train/tensor_param_model_layers_17_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_88_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_55_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/std":6.97586524536306e-05,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/max_abs":0.00022125244140625,"train/train/tensor_act_model_layers_63_mlp_down_proj/std":0.11413755383558184,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp_up_proj/std":0.38183724483646675,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/norm":0.021134122430673294,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/std":1.0067769242258642e-05,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/norm":5.15625,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/std":0.045166015625,"train/train/tensor_act_model_layers_76/mean":0.0767822265625,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/norm":6.875,"train/train/layer_model_layers_55/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn/mean":0.001895904541015625,"train/train/tensor_act_model_layers_43_self_attn_o_proj/std":0.10986562729861304,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_13/param/mean":0.0016655170601355305,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_91/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/norm":0.011540729673454776,"train/train/tensor_act_model_embed_tokens/std":0.11083985246422497,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_89/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/mean":-4.139728844165802e-07,"train/train/tensor_act_model_layers_0_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/std":0.05029296875,"train/train/layer_model_layers_88/grad/norm":0.10718360615419528,"train/train/tensor_act_model_layers_20_mlp_up_proj/max_abs":2.203125,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_v_proj/mean":0.00222015380859375,"train/train/tensor_act_model_layers_21_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/std":2.0238438819500053e-05,"train/train/layer_model_layers_72/act/max_abs":13.25,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/max_abs":0.000675201416015625,"train/train/tensor_act_model_layers_78_self_attn_k_proj/max_abs":5.625,"train/train/layer__model_layers_47/param/std":0.052747221489465916,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/max_abs":0.000720977783203125,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/mean":2.507586032152176e-07,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/max_abs":0.228515625,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/max_abs":0.000392913818359375,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/norm":4.03125,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/max_abs":0.1552734375,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/std":0.06005859375,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/std":3.7975256859286954e-05,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/mean":-3.618188202381134e-07,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/max_abs":0.00133514404296875,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_post_attention_layernorm/mean":0.0633544921875,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/mean":-0.00017261505126953125,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/std":0.0478515625,"train/train/tensor_param_model_layers_29_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_post_attention_layernorm/std":1.0000003948806937,"train/train/tensor_param_model_layers_89_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/mean":-0.0002536773681640625,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/mean":-4.9389200285077095e-08,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp_down_proj/max_abs":2.578125,"train/train/tensor_act_model_layers_48_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/norm":7.53125,"train/train/tensor_act_model_layers_87_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/norm":0.02624951702763524,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn/std":0.10974200919924976,"train/train/tensor_act_model_layers_68_post_attention_layernorm/std":1.0000012964001836,"train/train/tensor_act_model_layers_56_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/mean":-3.083987394347787e-07,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_5/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/norm":7.4375,"train/train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp/std":0.08386328889582995,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/max_abs":0.2392578125,"train/train/tensor_act_model_layers_15_self_attn_o_proj/norm":264.3069993316272,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/std":2.7441587797995977e-05,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/std":0.0537109375,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/max_abs":0.000461578369140625,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/norm":0.0393100523800769,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/norm":6.65625,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/mean":-1.0011353879235685e-07,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_60/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/norm":0.05164685141096109,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/norm":4.71875,"train/train/tensor_act_model_layers_19_self_attn/max_abs":0.76953125,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/norm":0.02942033819802947,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/std":1.749843319901735e-05,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/std":0.0380859375,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/max_abs":0.00063323974609375,"train/train/tensor_act_model_layers_48_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/max_abs":0.22265625,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/max_abs":0.18359375,"train/train/tensor_act_model_layers_86_input_layernorm/std":0.9960961659720049,"train/train/layer_model_layers_11/act/std":0.7041437194801465,"train/train/tensor_param_model_layers_51_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/max_abs":0.001129150390625,"train/train/tensor_act_model_layers_46_mlp_down_proj/mean":0.00310516357421875,"train/train/tensor_act_model_layers_44_self_attn_k_proj/mean":-0.0052490234375,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/max_abs":0.000240325927734375,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/max_abs":0.000431060791015625,"train/train/tensor_param_model_layers_43_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/max_abs":0.1259765625,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/std":0.031982421875,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/std":7.940865990101921e-05,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/mean":-3.039836883544922e-05,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_k_proj/mean":-0.0281982421875,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/std":0.038818359375,"train/train/tensor_act_model_layers_58_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_self_attn/max_abs":1.2421875,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/norm":0.01791951025716165,"train/train/tensor_act_model_layers_17_mlp_down_proj/norm":316.9139937447785,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/norm":0.02515070611273432,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/std":0.030029296875,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/mean":-0.0002498626708984375,"train/train/tensor_act_model_layers_6/norm":8623.977556193917,"train/train/layer__model_layers_45/param/std":0.05211408237826137,"train/train/tensor_act_model_layers_18_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_45/act/norm":14047.177743963912,"train/train/layer_model_layers_28/act/max_abs":17.875,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_v_proj/std":0.5117188830298149,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/norm":0.0009129129273930851,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/norm":0.014199498211866623,"train/train/tensor_act_model_layers_42_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/mean":0.00022411346435546875,"train/train/tensor_act_model_layers_73_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_26/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/mean":-2.726912498474121e-06,"train/train/tensor_act_model_layers_72_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/std":0.0001258577715865638,"train/train/tensor_act_model_layers_88_input_layernorm/max_abs":5.375,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/max_abs":0.00020694732666015625,"train/train/layer_model_layers_29/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_32/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/max_abs":0.2490234375,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/std":0.02734375,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/mean":-2.3923348635435104e-08,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/mean":8.255243301391602e-06,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/std":6.542365300987726e-05,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/max_abs":0.00144195556640625,"train/train/tensor_act_model_layers_16_self_attn/norm":234.52950527272105,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/mean":-0.0002593994140625,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/std":3.1994634938841595e-05,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/mean":-5.470588803291321e-06,"train/train/tensor_act_model_layers_19_self_attn_q_proj/std":0.8349630523119462,"train/train/tensor_act_model_layers_81_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp_up_proj/std":0.293458529197183,"train/train/tensor_act_model_layers_8_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/norm":7543.346431108587,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/norm":0.04980618343724597,"train/train/tensor_act_model_layers_7_self_attn/std":0.10046415939264479,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_mlp_down_proj/norm":807.1160007737263,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/norm":0.01414916828949647,"train/train/tensor_act_model_layers_42/max_abs":16.75,"train/train/tensor_act_model_layers_75/norm":9102.075650816458,"train/train/layer__model_layers_8/param/norm":19.730508578087896,"train/train/tensor_act_model_layers_27_self_attn_o_proj/norm":488.02966710732926,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/max_abs":0.000518798828125,"train/train/tensor_act_model_layers_78/std":1.6386792952286557,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/max_abs":0.25390625,"train/train/tensor_act_model_layers_61_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/mean":1.3629323802888393e-07,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/max_abs":0.000926971435546875,"train/train/tensor_act_model_layers_1_mlp_up_proj/norm":3319.9881140528187,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/mean":-2.518296241760254e-06,"train/train/tensor_act_model_layers_68_input_layernorm/max_abs":5.90625,"train/train/tensor_act_model_layers_19_input_layernorm/max_abs":6.09375,"train/train/tensor_act_model_layers_60_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/mean":3.701448440551758e-05,"train/train/layer_model_layers_36/grad/mean":-1.0224766796818018e-06,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_11/act/mean":-0.020569251133845404,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/std":1.931705659845249e-05,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/max_abs":0.1357421875,"train/train/tensor_act_model_layers_62_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/norm":0.001276829949045535,"train/train/tensor_act_model_layers_39_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_q_proj/norm":5759.94411004832,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/max_abs":0.00052642822265625,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/std":6.798507506824472e-05,"train/train/tensor_act_model_layers_62_mlp_up_proj/norm":4613.258476313635,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/max_abs":0.0009918212890625,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/std":0.055419921875,"train/train/tensor_act_model_layers_76_self_attn_q_proj/norm":5479.880701596757,"train/train/tensor_act_model_layers_43_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/std":0.02294921875,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/mean":-4.076957702636719e-05,"train/train/tensor_act_model_layers_58_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/max_abs":7.390975952148438e-05,"train/train/tensor_act_model_layers_13_input_layernorm/mean":0.001750946044921875,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/mean":9.755603969097137e-08,"train/train/tensor_act_model_layers_60_self_attn_k_proj/mean":0.0098876953125,"train/train/tensor_act_model_layers_65_post_attention_layernorm/std":1.0000004023312714,"train/train/tensor_act_model_layers_23_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/std":6.581571406939649e-05,"train/train/tensor_act_model_layers_39_input_layernorm/mean":0.0753173828125,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_66/grad/mean":2.543919962393512e-08,"train/train/tensor_act_model_layers_46_mlp_up_proj/std":0.35595803594251213,"train/train/tensor_act_model_layers_78_self_attn_o_proj/norm":1001.9470209422302,"train/train/tensor_param_model_layers_49_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_90_self_attn_o_proj/norm":898.9729478617083,"train/train/tensor_act_model_layers_31_post_attention_layernorm/norm":5792.609375006822,"train/train/tensor_act_model_layers_16_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_mlp_up_proj/max_abs":3.859375,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/mean":0.00054168701171875,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/std":0.00018002076248542265,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/max_abs":0.00010585784912109375,"train/train/tensor_act_model_layers_18_self_attn_v_proj/max_abs":2.171875,"train/train/tensor_act_model_layers_62_self_attn_q_proj/mean":0.13671875,"train/train/tensor_act_model_layers_64_mlp_down_proj/max_abs":1.2890625,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_q_proj/max_abs":5.40625,"train/train/tensor_act_model_layers_1_self_attn_k_proj/max_abs":5.03125,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29/std":1.3203240738310646,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/std":0.00018651553778573765,"train/train/tensor_act_model_layers_39_mlp_up_proj/std":0.3042014475537205,"train/train/tensor_act_model_layers_13_self_attn_v_proj/mean":0.0011768341064453125,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/mean":7.659196853637695e-06,"train/train/tensor_act_model_layers_60_mlp_up_proj/max_abs":3.234375,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/std":0.03955078125,"train/train/tensor_act_model_layers_77_mlp/mean":-0.010528564453125,"train/train/tensor_act_model_layers_49_mlp_up_proj/std":0.37060667497642147,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_post_attention_layernorm/mean":0.0440673828125,"train/train/tensor_act_model_layers_42_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/std":4.645601616318556e-05,"train/train/tensor_param_model_layers_60_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_v_proj/max_abs":3.078125,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn_k_proj/mean":-0.04736328125,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/max_abs":0.00022983551025390625,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/norm":0.015412736587224545,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/max_abs":0.000507354736328125,"train/train/tensor_act_model_layers_26_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/norm":9.4375,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_24_self_attn_o_proj/norm":447.342410913418,"train/train/tensor_act_model_layers_30_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/mean":0.0003795623779296875,"train/train/tensor_act_model_layers_29_input_layernorm/mean":0.050537109375,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/norm":5.28125,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp_down_proj/norm":3035.3361401297775,"train/train/layer_model_layers_78/act/std":0.7633982947745903,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/max_abs":0.2021484375,"train/train/layer__model_layers_92/param/mean":0.0014270776519537344,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/mean":2.641230821609497e-06,"train/train/layer_model_layers_28/grad/std":4.8926345719959603e-05,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/std":3.512434265319567e-05,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/std":0.037353515625,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/norm":0.01055722964581773,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_mlp_up_proj/frac_near_user_limit":0,"train/train/global/grad/std":9.170586221122621e-05,"train/train/tensor_act_model_layers_90_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_v_proj/norm":2387.1694455165884,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/max_abs":0.000293731689453125,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/max_abs":0.00018787384033203125,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/max_abs":0.00025177001953125,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/mean":4.5693013817071915e-08,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_29_mlp_up_proj/norm":2966.28657577892,"train/train/tensor_act_model_layers_5_post_attention_layernorm/mean":0.003437042236328125,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/max_abs":0.0004215240478515625,"train/train/tensor_act_model_layers_36_self_attn_o_proj/max_abs":1.1015625,"train/train/layer_model_layers_34/act/max_abs":17.875,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/std":1.727542863987636e-05,"train/train/tensor_act_model_layers_69_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_input_layernorm/std":1.0000001341104419,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/mean":-2.412707544863224e-08,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/norm":5.78125,"train/train/tensor_act_model_layers_56_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_up_proj/mean":-0.145751953125,"train/train/tensor_act_model_layers_30/max_abs":17.75,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/std":3.8975709811878056e-05,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_input_layernorm/max_abs":5.84375,"train/train/tensor_act_model_layers_69_self_attn_o_proj/std":0.13605561299317123,"train/train/tensor_act_model_layers_55_input_layernorm/std":1.0000001396983766,"train/train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp_up_proj/std":0.3261720291868291,"train/train/tensor_act_model_layers_53_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_30/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/norm":0.012499211703716404,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/max_abs":0.16796875,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/norm":0.0010408113303309653,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/max_abs":0.000713348388671875,"train/train/tensor_act_model_layers_75_mlp_up_proj/norm":5038.316607334216,"train/train/tensor_act_model_layers_10_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp/norm":1065.8288129052837,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/std":1.0234837156087998e-05,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/max_abs":0.25390625,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/std":0.0419921875,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_o_proj/mean":-0.001678466796875,"train/train/tensor_act_model_layers_47_mlp_down_proj/frac_near_user_limit":0,"train/train/layer__model_layers_93/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/mean":-7.390975952148438e-05,"train/train/tensor_act_model_layers_45_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_o_proj/norm":356.14257127885116,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/mean":-1.4295801520347595e-07,"train/train/tensor_act_model_layers_22_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp_down_proj/std":0.07751495894422442,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/std":0.041015625,"train/train/tensor_act_model_layers_30_mlp_down_proj/norm":264.15771479480543,"train/train/tensor_param_model_layers_18_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/max_abs":0.0001430511474609375,"train/train/tensor_act_model_layers_45_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_mlp_up_proj/norm":2479.5212425043474,"train/train/tensor_param_model_layers_72_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_44_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/norm":5.0625,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/norm":0.0076027978793013875,"train/train/tensor_act_model_layers_45_mlp/max_abs":0.82421875,"train/train/tensor_param_model_layers_86_input_layernorm_weight/std":0,"train/train/layer_model_layers_48/grad/max_abs":0.00130462646484375,"train/train/tensor_act_model_layers_52_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_30/act/std":0.6936556090270022,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/mean":-1.1026859283447266e-05,"train/train/layer__model_layers_8/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/mean":-9.648501873016357e-07,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/max_abs":0.283203125,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/std":0.0001167886230498691,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/norm":0.06012269607697877,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp_up_proj/max_abs":3.453125,"train/train/tensor_param_model_layers_21_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_up_proj/norm":5073.348514826938,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/max_abs":0.08935546875,"train/train/tensor_act_model_layers_82_self_attn_k_proj/max_abs":6.03125,"train/train/tensor_act_model_layers_77_self_attn_o_proj/max_abs":2.203125,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/max_abs":0.00016880035400390625,"train/train/tensor_act_model_layers_85_self_attn_q_proj/std":0.9882818884527081,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/std":0.040771484375,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_q_proj/max_abs":6.28125,"train/train/tensor_act_model_layers_8_self_attn_k_proj/mean":0.03302001953125,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/norm":0.0013462206633594793,"train/train/tensor_param_model_layers_75_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_post_attention_layernorm/max_abs":5.96875,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/norm":3.734375,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/mean":-1.4528632164001465e-07,"train/train/tensor_act_model_layers_80_self_attn_q_proj/mean":0.0709228515625,"train/train/tensor_act_model_layers_46_self_attn_o_proj/std":0.10180817654077728,"train/train/tensor_act_model_layers_17_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_83/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/max_abs":0.0007781982421875,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/max_abs":0.000591278076171875,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_81_self_attn_o_proj/mean":-0.002521514892578125,"train/train/tensor_param_model_layers_2_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_input_layernorm/mean":0.04461669921875,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/mean":-0.00017642974853515625,"train/train/tensor_act_model_layers_33_self_attn_o_proj/max_abs":0.671875,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_k_proj/max_abs":5.78125,"train/train/tensor_act_model_layers_60_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model/norm":5792.611938479005,"train/train/tensor_act_model_layers_79_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/mean":5.559995770454407e-07,"train/train/tensor_act_model_layers_11_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_30/grad/mean":-5.826892197937029e-07,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_37_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp_down_proj/mean":0.017547607421875,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/max_abs":0.000732421875,"train/train/tensor_act_model_layers_12_self_attn_k_proj/norm":4618.99982633296,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/max_abs":0.004913330078125,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/max_abs":0.00022602081298828125,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/mean":-8.0108642578125e-05,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/norm":6.625,"train/train/tensor_act_model_layers_54_self_attn_q_proj/max_abs":5.84375,"train/train/tensor_act_model_layers_44_input_layernorm/max_abs":5.90625,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/max_abs":0.00121307373046875,"train/train/tensor_act_model_layers_56/std":1.3183755308503544,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44/mean":0.08056640625,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_q_proj/norm":6118.714515571911,"train/train/tensor_act_model_layers_44/norm":7553.366303905261,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/norm":4.4375,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/norm":0.0020414155586503447,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/mean":-3.408058546483517e-08,"train/train/tensor_act_model_layers_40_self_attn_k_proj/max_abs":5.34375,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/norm":0.019660332418039408,"train/train/tensor_act_model_layers_61_self_attn_q_proj/max_abs":5.5625,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_15/param/max_abs":1,"train/train/layer_model_layers_59/grad/std":5.992086433161146e-05,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/std":2.6965238936859176e-05,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/mean":-9.371433407068253e-08,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/std":3.937149338833177e-05,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/max_abs":0.000431060791015625,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/norm":5.0625,"train/train/tensor_act_model_layers_70_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36/mean":0.0908203125,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/norm":0.00364149969975842,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_mlp_down_proj/norm":790.9359569796349,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_30_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_post_attention_layernorm/std":1.0000000223517416,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/mean":-0.0002803802490234375,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_49/param/mean":0.0011413859874707488,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/norm":0.007206576004743504,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_39/act/norm":13768.715504684154,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/norm":0.08399147181134603,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/mean":-3.01864929497242e-07,"train/train/tensor_act_model_layers_14_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88/std":2.105483400085098,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_17/param/mean":0.0016525316164013749,"train/train/tensor_act_model_layers_13_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn_v_proj/mean":0.0005841255187988281,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_self_attn_o_proj/mean":-0.0004942417144775391,"train/train/tensor_act_model_layers_29_self_attn_o_proj/norm":303.15854645793576,"train/train/tensor_act_model_layers_9_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/mean":0.000244140625,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_92/param/norm":25.80610544280171,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/max_abs":0.0012969970703125,"train/train/tensor_act_model_layers_10_self_attn_o_proj/mean":-0.000850677490234375,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/std":0.039794921875,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/mean":-2.6345252990722656e-05,"train/train/tensor_act_model_layers_24_post_attention_layernorm/norm":5792.605957035479,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/std":2.5187276992232653e-05,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/norm":6.5625,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/max_abs":0.000682830810546875,"train/train/tensor_act_model_layers_16_self_attn_v_proj/norm":1473.4377214579501,"train/train/layer_model_layers_61/act/std":0.7014200944206654,"train/train/tensor_act_model_layers_34_self_attn_o_proj/norm":502.29755068105896,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_post_attention_layernorm/norm":5792.609863288041,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/max_abs":0.1455078125,"train/train/tensor_act_model_layers_28_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/std":5.003088017089771e-05,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_mlp_down_proj/std":0.04895035049242646,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_input_layernorm/norm":5792.610473635482,"train/train/tensor_act_model_layers_38_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/max_abs":0.1259765625,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/std":3.293937674956196e-05,"train/train/tensor_act_model_layers_56_self_attn/mean":0.00016689300537109375,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/std":1.6012709353664897e-05,"train/train/tensor_act_model_layers_1_mlp_down_proj/mean":-0.00711822509765625,"train/train/tensor_act_model_layers_55/mean":0.06011962890625,"train/train/tensor_act_model_layers_9_post_attention_layernorm/std":1.0000000057043505,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_k_proj/std":1.0000001629814372,"train/train/layer_model_layers_87/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_42_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_34_post_attention_layernorm/mean":0.0650634765625,"train/train/tensor_act_model_layers_64_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/max_abs":0.1318359375,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/norm":0.006320933800657252,"train/train/layer__model_layers_26/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/max_abs":0.000392913818359375,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/max_abs":0.000553131103515625,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/norm":5.65625,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/norm":0.0003448373446684731,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/norm":5.625,"train/train/layer_model_layers_56/act/mean":-0.012281417846679688,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/max_abs":0.000514984130859375,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/mean":0.00021457672119140625,"train/train/tensor_act_model_layers_29/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/mean":-0.000217437744140625,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/mean":5.844049155712128e-08,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/max_abs":0.0004100799560546875,"train/train/tensor_act_model_layers_73_mlp/mean":0.010162353515625,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_o_proj/mean":0.0007352828979492188,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/norm":0.013863783405116844,"train/train/tensor_act_model_layers_36_self_attn_q_proj/max_abs":6,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/mean":0.00014400482177734375,"train/train/layer_model_layers_91/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn/max_abs":2.25,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/norm":6.1875,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/max_abs":0.0001621246337890625,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_input_layernorm/max_abs":6.1875,"train/train/tensor_act_model_layers_68_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_23/act/std":0.6874699153673076,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/norm":6.875,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/std":4.888764271740781e-05,"train/train/layer__model_layers_35/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/max_abs":0.0006561279296875,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/std":3.1581418425563235e-05,"train/train/tensor_act_model_layers_24_self_attn/norm":447.342410913418,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/max_abs":0.000701904296875,"train/train/tensor_act_model_layers_38_self_attn_o_proj/std":0.1242678663699174,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/max_abs":0.0005645751953125,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_55_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_down_proj/max_abs":0.466796875,"train/train/layer__model_layers_9/param/std":0.04958668254271731,"train/train/tensor_act_model_layers_78/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/mean":7.424387149512768e-08,"train/train/tensor_act_model_layers_28_self_attn_q_proj/std":1.0351639561771158,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/mean":-4.4226646423339844e-05,"train/train/tensor_act_model_layers_74/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn_k_proj/std":0.8056676506193773,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/mean":-1.1496245861053467e-05,"train/train/tensor_act_model_layers_6_self_attn_o_proj/std":0.09692538284413871,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/norm":6.125,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/std":2.2938949120562467e-05,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_82/grad/mean":1.5255485119388182e-06,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_15/std":1.3945586591853272,"train/train/tensor_act_model_layers_16_mlp_up_proj/max_abs":2.3125,"train/train/layer_model_layers_89/act/mean":-0.017697847806490384,"train/train/tensor_act_model_layers_40_self_attn_v_proj/mean":0.004150390625,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/std":6.318766355639779e-05,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/max_abs":0.0007781982421875,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/norm":0.011600453615979375,"train/train/tensor_act_model_layers_87_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_act_model_layers_11_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/norm":0.0004918602445474737,"train/train/tensor_param_model_layers_18_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/mean":-0.00017070770263671875,"train/train/layer_model_layers_40/grad/frac_near_user_limit":0,"train/train/layer_model_layers_31/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_post_attention_layernorm/mean":0.059814453125,"train/train/layer_model_layers_6/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/norm":0.011902520576868697,"train/train/tensor_act_model_layers_60_self_attn_o_proj/mean":-0.003444671630859375,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/std":5.503484055503452e-05,"train/train/tensor_act_model_layers_60_self_attn/max_abs":1.53125,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/std":2.1859092539726697e-05,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_o_proj/max_abs":2.296875,"train/train/tensor_act_model_layers_77_self_attn_q_proj/max_abs":6.8125,"train/train/tensor_act_model_layers_41_self_attn_v_proj/mean":-0.003292083740234375,"train/train/tensor_param_model_layers_93_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_91/param/norm":25.383657697916586,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/std":2.1637578982934116e-05,"train/train/tensor_act_model_layers_67_self_attn_o_proj/std":0.09510203024815737,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/std":0.051513671875,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/max_abs":0.000591278076171875,"train/train/tensor_act_model_layers_20_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/mean":-2.833257894963026e-08,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/mean":-0.0002651214599609375,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_79_self_attn_k_proj/max_abs":6,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/mean":-4.7206878662109375e-05,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/norm":0.022380435810763308,"train/train/layer_model_layers_45/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/max_abs":0.232421875,"train/train/tensor_act_model_layers_42_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_input_layernorm/mean":0.0714111328125,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn_v_proj/max_abs":2.6875,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/std":0.042724609375,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/max_abs":0.000438690185546875,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/norm":4.8125,"train/train/tensor_act_model_layers_22_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/mean":-1.0291114449501038e-06,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/mean":6.165355443954468e-07,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/mean":0.00060272216796875,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_12/act/norm":13742.376070422139,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_v_proj/norm":2091.758700408335,"train/train/tensor_act_model_layers_33_self_attn_q_proj/max_abs":5.34375,"train/train/tensor_act_model_layers_31_input_layernorm/std":1.0000000745058033,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn/norm":356.14257127885116,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/mean":-0.0005035400390625,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/mean":9.447336196899414e-06,"train/train/tensor_act_model_layers_77_post_attention_layernorm/std":1.000002210957348,"train/train/tensor_act_model_layers_15_input_layernorm/mean":0.0077362060546875,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/max_abs":0.125,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_1_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/max_abs":0.000530242919921875,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/max_abs":0.10205078125,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_45_self_attn_v_proj/norm":2115.149666191745,"train/train/layer_model_layers_0/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_24/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_o_proj/max_abs":1.2890625,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_57_mlp_up_proj/std":0.4121095196330463,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/norm":0.017327643295761107,"train/train/tensor_act_model_layers_82_mlp/max_abs":2.328125,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/std":1.2531047857719672e-05,"train/train/tensor_act_model_layers_46_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/mean":-9.298324584960938e-06,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/max_abs":9.107589721679688e-05,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/std":2.8717307292435082e-05,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/max_abs":0.10498046875,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/mean":0.0003814697265625,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/std":6.64933059304322e-05,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_o_proj/max_abs":1.0625,"train/train/layer__model_layers_67/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/max_abs":0.1474609375,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_mlp_down_proj/std":0.08349610438123714,"train/train/tensor_act_model_layers_31_self_attn_q_proj/norm":5617.224158444125,"train/train/tensor_param_model_layers_28_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/mean":3.1948089599609375e-05,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/norm":0.01860291087051208,"train/train/tensor_act_model_layers_13_self_attn_k_proj/norm":5421.069827422131,"train/train/tensor_act_model_layers_71/norm":8636.848977248768,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/std":0.0274658203125,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/max_abs":0.0001392364501953125,"train/train/tensor_param_model_layers_71_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_84/grad/norm":0.08014847228486574,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/std":0.034423828125,"train/train/tensor_param_model_layers_32_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_input_layernorm/std":1.0000000055879354,"train/train/layer_model_layers_81/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp/norm":573.196105520505,"train/train/tensor_act_model_layers_85_self_attn_q_proj/norm":5726.338433523564,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/mean":0.00063323974609375,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/std":2.313444151793943e-05,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/norm":0.003281304935154402,"train/train/layer_model_layers_38/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/norm":7.03125,"train/train/tensor_act_model_layers_68_self_attn_q_proj/mean":0.00717926025390625,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/max_abs":0.2216796875,"train/train/tensor_act_model_layers_1_mlp_down_proj/max_abs":2.046875,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_input_layernorm/std":1.0000000074505806,"train/train/tensor_act_model_layers_51_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_25/grad/norm":0.036125284716454174,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/norm":0.026435903794099005,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/std":0.027587890625,"train/train/tensor_param_model_layers_70_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/norm":0.012558011209079359,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/mean":9.775161743164062e-05,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_42/act/mean":-0.0044428018423227165,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/norm":0.02314689398829548,"train/train/tensor_act_model_layers_75_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/std":0.0380859375,"train/train/tensor_act_model_layers_37_post_attention_layernorm/max_abs":6.15625,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/max_abs":0.000152587890625,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/norm":0.0010512934707275902,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/max_abs":0.0020294189453125,"train/train/tensor_param_model_layers_45_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_34/grad/max_abs":0.0025634765625,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_92/grad/max_abs":0.00135040283203125,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_q_proj/max_abs":6.84375,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_54_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_self_attn/max_abs":0.8515625,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/mean":-3.7513673305511475e-06,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/norm":10.5625,"train/train/tensor_act_model_layers_74_self_attn_o_proj/mean":-0.0012788772583007812,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/std":0.00022177691710581758,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/mean":-5.362089723348618e-07,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/norm":0.02033670890127608,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_post_attention_layernorm/norm":5792.607666022771,"train/train/tensor_act_model_layers_60_post_attention_layernorm/max_abs":5.9375,"train/train/tensor_act_model_layers_3_self_attn/frac_near_dtype_limit":0,"train/train/layer__model_layers_71/param/max_abs":1,"train/train/tensor_act_model_layers_38_mlp_down_proj/mean":-0.0092620849609375,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/std":0.04931640625,"train/train/tensor_act_model_layers_54_self_attn_k_proj/norm":4660.29093090709,"train/train/tensor_act_model_layers_12_self_attn_o_proj/std":0.026341273995517524,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/std":0.0244140625,"train/train/layer_model_layers_2/grad/std":6.024665606774132e-05,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/std":2.801731252381122e-05,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/mean":1.3085082173347473e-07,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/std":0.03564453125,"train/train/tensor_act_model_layers_50_self_attn_o_proj/std":0.05645898675300815,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_35/mean":0.08984375,"train/train/tensor_act_model_layers_52_mlp/norm":530.4828839925549,"train/train/tensor_act_model_layers_90_self_attn_k_proj/std":0.798831240174578,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_input_layernorm/std":1.0000001396983766,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/mean":9.103678166866302e-08,"train/train/tensor_act_model_layers_42_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/max_abs":0.25390625,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp_down_proj/mean":-0.003265380859375,"train/train/tensor_act_model_layers_51_post_attention_layernorm/std":1.0000001396983766,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/max_abs":0.0004634857177734375,"train/train/tensor_act_model_layers_28/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_80/grad/norm":0.07929659983820342,"train/train/tensor_act_model_layers_90_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/std":6.724127437914975e-05,"train/train/layer__model_layers_51/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/mean":0.00019073486328125,"train/train/tensor_act_model_layers_76/max_abs":12.625,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/std":0.0228271484375,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_75_mlp_down_proj/max_abs":1.453125,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/norm":0.005998043464006294,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_q_proj/mean":-0.0164794921875,"train/train/tensor_act_model_layers_38_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/norm":0.0006075616492508129,"train/train/layer_model_layers_93/grad/norm":0.10333069736633875,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_o_proj/std":0.049256506181702756,"train/train/tensor_act_model_layers_92_mlp_down_proj/max_abs":10.8125,"train/train/tensor_act_model_layers_68_mlp_up_proj/std":0.4848640802997255,"train/train/tensor_act_model_rotary_emb/std":0.71875,"train/train/tensor_param_model_layers_2_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_48_input_layernorm/max_abs":5.65625,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/mean":0.000186920166015625,"train/train/layer__model_layers_79/param/norm":23.697474812730576,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/norm":0.0020414191228199387,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/max_abs":0.1455078125,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/norm":6.78125,"train/train/layer_model_layers_40/act/mean":-0.02651361318734976,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/max_abs":0.000560760498046875,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_mlp_up_proj/norm":2190.011939361449,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/norm":0.0005977461422374096,"train/train/layer_model_layers_90/grad/mean":3.783566973138711e-07,"train/train/layer__model_layers_29/param/mean":0.0018924231090337363,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/norm":5.65625,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/mean":4.090368747711182e-06,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/mean":1.1676456779241562e-07,"train/train/tensor_act_model_layers_18_self_attn_q_proj/norm":5675.660678434185,"train/train/tensor_act_model_layers_86_mlp_up_proj/max_abs":4.21875,"train/train/tensor_act_model_layers_29_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_up_proj/norm":2806.5418562940167,"train/train/tensor_act_model_layers_24/std":1.3437613885973552,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_43/param/norm":21.413912862027342,"train/train/tensor_act_model_layers_35_self_attn_v_proj/norm":2001.7767688469999,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/norm":2.828125,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/std":0.052734375,"train/train/tensor_act_model_layers_48_post_attention_layernorm/max_abs":5.78125,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn/mean":0.0019779205322265625,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/max_abs":0.263671875,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/norm":0.02417300426210077,"train/train/tensor_act_model_layers_47_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/norm":4.34375,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/max_abs":0.0009918212890625,"train/train/tensor_act_model_layers_81_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_input_layernorm/mean":0.0589599609375,"train/train/tensor_act_model_layers_89_self_attn_k_proj/mean":0.0124359130859375,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_mlp/mean":0.0070953369140625,"train/train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/std":7.212447240995534e-05,"train/train/layer__model_layers_82/param/std":0.05864702842078529,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/std":0.04443359375,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_post_attention_layernorm/std":1.000002544370037,"train/train/tensor_act_model_layers_68_self_attn/norm":1043.7661422415608,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/std":8.895666749033523e-05,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_post_attention_layernorm/std":0.997071991936804,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/mean":6.723403930664062e-05,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/mean":-1.1066440492868423e-06,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_56_mlp_down_proj/std":0.08386328889582995,"train/train/tensor_act_model_layers_2_self_attn/norm":260.3362169687711,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/std":6.039388740168621e-05,"train/train/layer_model_layers_73/act/std":0.7085335046655846,"train/train/tensor_act_model_layers_62_input_layernorm/mean":0.05926513671875,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/max_abs":0.0003528594970703125,"train/train/layer__model_layers_70/param/std":0.06223362508540903,"train/train/tensor_act_model_layers_75_self_attn_v_proj/mean":-0.00014281272888183594,"train/train/layer_model_layers_35/grad/mean":-7.583261373076535e-07,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/norm":0.015155196448213578,"train/train/layer_model_layers_62/act/mean":0.00876602759728065,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/mean":3.978610038757324e-06,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/std":2.8657070809224807e-05,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn/mean":-0.002048492431640625,"train/train/tensor_act_model_layers_58_self_attn_k_proj/norm":4305.172526458116,"train/train/layer_model_layers_87/act/norm":18561.072270832923,"train/train/tensor_act_model_layers_5_mlp/std":0.05664062808299879,"train/train/layer__model_layers_39/param/std":0.050914773666392334,"train/train/tensor_act_model_layers_72_self_attn_k_proj/mean":0.05877685546875,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/mean":0.00022602081298828125,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn/norm":1231.0956781488853,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/mean":7.104873657226562e-05,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/mean":-2.905726432800293e-06,"train/train/tensor_act_model_layers_48/max_abs":15.75,"train/train/tensor_act_model_layers_19/std":1.3633092587717142,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/std":3.1614858892098915e-05,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67/mean":0.0562744140625,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/norm":0.03447184359103202,"train/train/tensor_act_model_layers_69_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/max_abs":0.1357421875,"train/train/tensor_act_model_layers_9_self_attn_q_proj/mean":-0.0911865234375,"train/train/tensor_act_model_layers_46_self_attn/max_abs":1.375,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/norm":0.005872354436814473,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/max_abs":0.00022125244140625,"train/train/tensor_act_model_layers_48_mlp_down_proj/max_abs":1.2265625,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83/mean":0.1280517578125,"train/train/tensor_act_model_layers_88_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_0/grad/norm":0.4506515401639793,"train/train/tensor_param_model_layers_84_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp/mean":-0.0084686279296875,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/std":5.261379552443366e-05,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/mean":7.613562047481537e-08,"train/train/tensor_act_model_layers_68_self_attn_v_proj/std":0.46386824253630493,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/max_abs":0.000682830810546875,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/norm":0.02434292920923935,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/std":0.03759765625,"train/train/tensor_act_model/mean":0.058349609375,"train/train/tensor_act_model_layers_92_post_attention_layernorm/std":1.000001093372105,"train/train/tensor_act_model_layers_57/norm":7663.320685441612,"train/train/layer__model_layers_38/param/max_abs":1,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/norm":5.96875,"train/train/tensor_act_model_layers_8_input_layernorm/mean":-0.00707244873046875,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/max_abs":0.1708984375,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/mean":-1.7490237951278687e-06,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/epoch":0.8088433540037746,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/norm":5.9375,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_v_proj/max_abs":1.59375,"train/train/tensor_act_model_layers_39_mlp_up_proj/max_abs":2.671875,"train/train/tensor_act_model_layers_46_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/mean":0.0001850128173828125,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/std":0.022216796875,"train/train/tensor_act_model_layers_19_self_attn_o_proj/max_abs":0.76953125,"train/train/tensor_act_model_layers_55_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/std":0.044189453125,"train/train/tensor_act_model_layers_70_self_attn_o_proj/norm":1506.1802860033656,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/std":0.040283203125,"train/train/tensor_act_model_layers_20_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_65/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/norm":5.625,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/norm":5.375,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/mean":-0.0003566741943359375,"train/train/layer_model_layers_45/grad/std":5.3753671623839726e-05,"train/train/tensor_act_model_layers_52_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/std":6.831839268211372e-05,"train/train/tensor_act_model_layers_89_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_q_proj/norm":5541.052115882187,"train/train/tensor_param_model_layers_13_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/max_abs":6.9375,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_up_proj/std":0.2346195216974663,"train/train/tensor_act_model_layers_45_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/std":0.031494140625,"train/train/tensor_act_model_layers_78_input_layernorm/std":1.0000022109573479,"train/train/tensor_act_model_layers_26_self_attn_v_proj/max_abs":2.078125,"train/train/tensor_act_model_layers_34_input_layernorm/norm":5792.604980472588,"train/train/tensor_param_model_layers_1_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_15_post_attention_layernorm/std":1.0000000487780187,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/std":0.043701171875,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/std":0.0517578125,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/mean":-1.8905848264694214e-06,"train/train/tensor_act_model_layers_45_input_layernorm/max_abs":5.875,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/std":0.029541015625,"train/train/tensor_act_model_layers_29_self_attn_q_proj/norm":5617.403319114031,"train/train/layer_model_layers_55/act/mean":-0.014397841233473558,"train/train/tensor_act_model_layers_62_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/mean":7.747439667582512e-08,"train/train/tensor_act_model_layers_43_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn/max_abs":2.796875,"train/train/tensor_act_model_layers_49_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp_up_proj/max_abs":3.296875,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/std":0.034912109375,"train/train/layer__model_layers_88/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/std":0.060546875,"train/train/tensor_act_model_layers_91_mlp/norm":3525.8202406809833,"train/train/tensor_act_model_layers_50_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/mean":-1.1507654562592506e-07,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_84_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/mean":5.289912223815918e-06,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/mean":5.304813385009766e-06,"train/train/layer__model_layers_79/param/mean":0.0011414335967969969,"train/train/tensor_act_model_layers_5/max_abs":18.75,"train/train/tensor_param_model_layers_83_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/mean":-1.7452111933380365e-07,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp/std":0.5224637040394773,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_56/grad/norm":0.041089666080301526,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/norm":0.017031105460857246,"train/train/layer_model_layers_54/act/std":0.6674520238578646,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/norm":3.4375,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/norm":3.5,"train/train/layer_model_layers_45/grad/mean":-5.703340989193195e-07,"train/train/tensor_act_model_layers_2_self_attn_v_proj/max_abs":1.609375,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/mean":-1.0165804269490764e-06,"train/train/tensor_act_model_layers_28_self_attn/mean":-0.001529693603515625,"train/train/tensor_act_model_layers_56_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_k_proj/std":0.8691434238688804,"train/train/tensor_act_model_layers_49_self_attn_q_proj/std":0.9492191326470252,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85/norm":11053.738989594316,"train/train/tensor_act_model_layers_24_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_k_proj/norm":5002.1388573914555,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/norm":3.515625,"train/train/global/param/mean":0.0011642012166190522,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/mean":2.852175384759903e-08,"train/train/layer_model_layers_13/act/max_abs":18.125,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62/norm":7951.49988722598,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/std":0.048828125,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/std":0.0223388671875,"train/train/tensor_act_model_layers_37_post_attention_layernorm/mean":0.0787353515625,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_mlp_down_proj/std":0.12744240348204688,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/mean":9.569339454174042e-08,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_51/act/norm":13799.782050731672,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_31/param/norm":20.274303028248593,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/std":2.1259261059914982e-05,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/norm":0.04677794262974541,"train/train/tensor_act_model_layers_36_mlp/std":0.06530795988545568,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/max_abs":0.1279296875,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/mean":6.565824151039124e-08,"train/train/tensor_act_model_layers_5_self_attn_q_proj/max_abs":6.375,"train/train/tensor_act_model_layers_92_self_attn_o_proj/mean":-0.006683349609375,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/mean":-4.330649971961975e-07,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn/std":0.044922308346353024,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/std":3.358082472603266e-05,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/std":0.044677734375,"train/train/tensor_act_model_layers_50/std":1.2929748393834484,"train/train/tensor_act_model_layers_92_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_rotary_emb/norm":2297.209228515625,"train/train/tensor_act_model_layers_27_post_attention_layernorm/max_abs":6.3125,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/norm":0.020572016893500435,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/mean":0.000118255615234375,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/norm":0.012515184001796825,"train/train/tensor_act_model_layers_21_self_attn_k_proj/norm":5633.60320954809,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/norm":7.34375,"train/train/tensor_act_model_layers_27_self_attn_v_proj/max_abs":2.21875,"train/train/layer_model_layers_36/act/mean":-0.0017990699181189905,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_17/grad/norm":0.03684007529430652,"train/train/tensor_act_model_layers_7_mlp_up_proj/max_abs":3.109375,"train/train/tensor_act_model_layers_75_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/std":1.8405143492503552e-05,"train/train/tensor_act_model_layers_86_post_attention_layernorm/norm":5792.613159181976,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/mean":1,"train/train/layer__model_layers_88/param/std":0.06121913213058473,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp_down_proj/max_abs":0.75390625,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/mean":-5.602836608886719e-05,"train/train/tensor_act_model_layers_11_mlp/max_abs":0.310546875,"train/train/layer_model_layers_67/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_88_self_attn_v_proj/max_abs":3.390625,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_84/act/norm":17232.725046214586,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/max_abs":0.00101470947265625,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/mean":1.4505349099636078e-07,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/std":5.948460680827795e-05,"train/train/tensor_act_model_layers_77_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/norm":0.0012978158498117326,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/std":0.0255126953125,"train/train/tensor_act_model_layers_80/std":1.724616853172877,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_k_proj/mean":0.03369140625,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/max_abs":0.1689453125,"train/train/tensor_act_model_layers_11_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_up_proj/std":0.44043090549478203,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn_k_proj/mean":-0.05548095703125,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_62/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/std":3.1520451735670236e-05,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/norm":0.0254340900178798,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/norm":0.017281897286128296,"train/train/tensor_param_model_layers_85_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp/max_abs":8.6875,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp_up_proj/max_abs":7.21875,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/max_abs":0.000278472900390625,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/mean":-0.001007080078125,"train/train/tensor_act_model_layers_45_input_layernorm/std":0.99609389959596,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/norm":4.96875,"train/train/tensor_act_model_layers_13_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/norm":0.013536016648814235,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/std":0.0576171875,"train/train/tensor_act_model_layers_27_self_attn/mean":0.0005831718444824219,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/std":0.05419921875,"train/train/tensor_act_model_layers_30_mlp_up_proj/max_abs":2.328125,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/max_abs":0.001495361328125,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/max_abs":0.1826171875,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/std":0.0264892578125,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/mean":-0.00045013427734375,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70/norm":8597.535086355512,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/std":8.956007682988442e-05,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/max_abs":0.1884765625,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/std":2.820094726369281e-05,"train/train/tensor_param_model_layers_38_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/norm":0.0066336784774747954,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/max_abs":0.119140625,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/std":0.0245361328125,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/mean":-0.00011444091796875,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/max_abs":0.0002841949462890625,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/max_abs":0.000244140625,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_71_self_attn/mean":-0.00051116943359375,"train/train/tensor_act_model_layers_79_self_attn_q_proj/std":1.0859378018824755,"train/train/tensor_act_model_layers_78_mlp_up_proj/max_abs":3.578125,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/std":0.04345703125,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/norm":0.028356923084035396,"train/train/tensor_act_model_layers_6_mlp_down_proj/mean":-0.00548553466796875,"train/train/tensor_act_model_layers_51_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn_q_proj/mean":0.0144500732421875,"train/train/tensor_param_model_layers_52_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_78_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_mlp_down_proj/std":0.04040545783332169,"train/train/tensor_act_model_layers_33_self_attn_k_proj/mean":-0.0263671875,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_mlp/std":0.08349610438123714,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_mlp_up_proj/mean":-0.05242919921875,"train/train/tensor_act_model_layers_85_self_attn_o_proj/norm":807.6520046408407,"train/train/tensor_param_model_layers_8_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/norm":0.014396244966728495,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/max_abs":0.0002613067626953125,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/mean":1.2665987014770508e-07,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/mean":-2.367887645959854e-07,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/norm":0.01554073020612559,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/norm":0.004167465540675382,"train/train/layer__model_layers_6/param/mean":0.0014594430670537369,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_13/act/mean":-0.005167374244103065,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/max_abs":0.00018405914306640625,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp/max_abs":0.7578125,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/norm":4.28125,"train/train/tensor_act_model_layers_16_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/norm":0.0006795463206429563,"train/train/tensor_act_model_layers_88_mlp_up_proj/max_abs":4.75,"train/train/tensor_act_model_layers_77_self_attn_k_proj/norm":5141.081119515607,"train/train/tensor_param_model_embed_tokens_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/std":3.191435799107909e-05,"train/train/tensor_act_model_layers_31_self_attn_o_proj/norm":294.3296823336847,"train/train/tensor_param_model_layers_71_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_76/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/max_abs":0.287109375,"train/train/tensor_act_model_layers_19_self_attn_q_proj/max_abs":4.90625,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/norm":0.010911303893144226,"train/train/tensor_act_model_layers_7_self_attn_o_proj/std":0.10046415939264479,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70/max_abs":13.4375,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/mean":-0.0002384185791015625,"train/train/layer_model_layers_70/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/norm":0.005670298235022018,"train/train/layer__model_layers_37/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/norm":8,"train/train/tensor_act_model_layers_76_mlp/mean":0.0020389556884765625,"train/train/layer_model_layers_65/grad/std":7.049803287173177e-05,"train/train/tensor_act_model_layers_60_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/norm":6.625,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp_up_proj/mean":-0.0926513671875,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/max_abs":0.0002880096435546875,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/max_abs":0.1962890625,"train/train/layer_model_layers_18/grad/mean":-5.473292689892505e-07,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/mean":-1.621432602405548e-05,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/norm":0.00309088420651219,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_post_attention_layernorm/norm":5792.613525392164,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/std":0.02490234375,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/max_abs":0.1806640625,"train/train/tensor_act_model_layers_33_self_attn/max_abs":0.671875,"train/train/tensor_act_model_layers_63_self_attn_o_proj/mean":-0.000804901123046875,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/norm":6.84375,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/std":0.034423828125,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/max_abs":0.0005645751953125,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/std":0.06884765625,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/std":0.04443359375,"train/train/tensor_act_model_layers_76_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79/mean":0.0960693359375,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/max_abs":0.2734375,"train/train/tensor_act_model_layers_46/max_abs":16.125,"train/train/layer_model_layers_45/grad/norm":0.04352222064334929,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn_v_proj/max_abs":5.5,"train/train/tensor_act_model_layers_60_self_attn_v_proj/std":0.4604501200508599,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_62/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp/mean":0.00466156005859375,"train/train/layer_model_layers_13/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/std":0.033935546875,"train/train/tensor_act_model_layers_42_input_layernorm/norm":5792.608520516902,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_o_proj/mean":0.0016231536865234375,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/max_abs":0.000278472900390625,"train/train/tensor_act_model_layers_83_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_o_proj/max_abs":0.91015625,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/norm":0.010135787744710988,"train/train/tensor_act_model_layers_12_mlp/norm":234.43811557411908,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/norm":4.40625,"train/train/tensor_act_model_layers_17_self_attn_k_proj/std":0.9921875826017089,"train/train/tensor_act_model_layers_54_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_48/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_up_proj/std":0.2324218990422084,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/std":1.155462864556188e-05,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/max_abs":0.14453125,"train/train/tensor_act_model_layers_35_self_attn_q_proj/mean":-0.023162841796875,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/mean":2.6437919586896896e-07,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/norm":0.0007380650598947845,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/std":0.02978515625,"train/train/tensor_act_model_layers_77_mlp/std":0.17285159213394513,"train/train/tensor_act_model_layers_53_self_attn_k_proj/norm":4338.042266747563,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/std":0.04833984375,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/std":3.040707484025692e-05,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/std":0.04296875,"train/train/tensor_act_model_layers_36_self_attn_k_proj/std":0.8828125617145416,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/max_abs":0.18359375,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_post_attention_layernorm/mean":0.03143310546875,"train/train/tensor_act_model_layers_88_self_attn_k_proj/std":1.1445381681864617,"train/train/tensor_act_model_layers_4_self_attn/max_abs":0.484375,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/mean":-1.8924474716186523e-06,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/std":1.2014072486063127e-05,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/mean":-0.000659942626953125,"train/train/layer_model_layers_55/grad/std":6.700485430898937e-05,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/mean":-1.671724021434784e-07,"train/train/tensor_act_model_layers_71_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_57/act/norm":14225.202897882744,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/max_abs":0.00018405914306640625,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/std":7.266812640311943e-05,"train/train/tensor_act_model_layers_37_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/mean":9.107589721679688e-05,"train/train/tensor_act_model_layers_78_mlp/max_abs":1.5,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/norm":5.8125,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/mean":1.0640360414981842e-07,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/norm":0.013826450095601909,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_v_proj/std":0.36914063568310745,"train/train/tensor_act_model_layers_48_post_attention_layernorm/norm":5792.608398441184,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_input_layernorm/std":1.0000001545995356,"train/train/tensor_param_model_layers_31_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_42_self_attn_v_proj/norm":1977.1744880752392,"train/train/layer_model_layers_86/grad/norm":0.08472549217958579,"train/train/tensor_act_model_layers_9_self_attn/std":0.18019368434800248,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/norm":0.028414667730884097,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/max_abs":0.000713348388671875,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/mean":1.294165849685669e-05,"train/train/tensor_act_model_layers_37_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/mean":4.939734935760498e-06,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/std":0.03515625,"train/train/tensor_act_model_layers_9_self_attn_o_proj/mean":-0.000759124755859375,"train/train/tensor_act_model_layers_48_self_attn_o_proj/mean":-0.00118255615234375,"train/train/tensor_act_model_layers_11_self_attn_v_proj/std":0.26855647809883865,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/std":0.00010436795506153836,"train/train/tensor_act_model_layers_18_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_mlp/mean":-0.00557708740234375,"train/train/tensor_act_model_layers_32/frac_near_user_limit":0,"train/train/layer_model_layers_79/grad/std":8.57656529400095e-05,"train/train/tensor_act_model_layers_1_post_attention_layernorm/norm":5792.60803222954,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/norm":0.0009337557650587452,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/max_abs":0.1572265625,"train/train/tensor_act_model_layers_68_self_attn/std":0.1801796702091887,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/mean":0.000370025634765625,"train/train/tensor_act_model_layers_72_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_self_attn_q_proj/std":0.8232439545672402,"train/train/tensor_act_model_layers_67_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/norm":0.026437101568551953,"train/train/layer_model_layers_31/act/mean":-0.004345527062049279,"train/train/tensor_act_model_layers_85_mlp_up_proj/std":0.5976567985183093,"train/train/tensor_act_model_layers_68_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/mean":5.476176738739014e-06,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_79_mlp/mean":0.0059051513671875,"train/train/tensor_act_model_layers_70_mlp_up_proj/std":0.5019569842119701,"train/train/tensor_act_model_layers_80_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/norm":0.012834773300507151,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/std":4.11914645543229e-05,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_32/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_40/param/std":0.05220513998228683,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/max_abs":0.1513671875,"train/train/tensor_act_model_layers_80_mlp_down_proj/max_abs":1.7265625,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/max_abs":0.00011348724365234375,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/max_abs":0.00034332275390625,"train/train/tensor_act_model_layers_72_mlp_up_proj/max_abs":3.140625,"train/train/tensor_act_model_layers_4/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_post_attention_layernorm/max_abs":5.9375,"train/train/layer_model_layers_86/act/norm":18001.65642864427,"train/train/tensor_act_model_layers_17_self_attn_o_proj/mean":-0.0008487701416015625,"train/train/tensor_act_model_layers_21_post_attention_layernorm/std":1.0000000353902572,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/norm":0.006057796910393555,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/mean":-5.740439519286156e-07,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_input_layernorm/std":0.996096165972005,"train/train/tensor_act_model_layers_50_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/std":0.0001667096220072103,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/std":7.364287441130947e-06,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/norm":0.029330107595979267,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp_up_proj/mean":-0.0604248046875,"train/train/tensor_act_model_layers_70_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_k_proj/mean":-0.024627685546875,"train/train/tensor_param_model_layers_1_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_80_mlp_down_proj/std":0.18457033063368733,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/std":0.041259765625,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/norm":0.020188248688363875,"train/train/layer__model_layers_63/param/mean":0.0011858411958548655,"train/train/layer__model_layers_1/param/norm":19.541997043227184,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_0/std":1.4629154623390728,"train/train/tensor_act_model_layers_11_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_2/act/mean":-0.013445047231820913,"train/train/tensor_act_model_layers_67_mlp_up_proj/std":0.47656312535979717,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/norm":8.0625,"train/train/tensor_act_model_layers_20_self_attn_q_proj/norm":4771.579896129194,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_post_attention_layernorm/mean":-0.0067901611328125,"train/train/tensor_act_model_layers_31_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_embed_tokens_weight/norm":0.1726659435468184,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_65_mlp_up_proj/mean":-0.0975341796875,"train/train/layer_model_layers_53/act/norm":13835.872946534337,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/std":1.2661691108044543e-05,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81/norm":10084.707088729754,"train/train/layer__model_layers_20/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_k_proj/norm":5209.181833312542,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/max_abs":0.000331878662109375,"train/train/layer_model_layers_71/grad/std":6.787615593491401e-05,"train/train/layer_model_layers_70/grad/max_abs":0.00160980224609375,"train/train/tensor_act_model_layers_10_mlp_up_proj/std":0.21777346315939713,"train/train/tensor_act_model_layers_34_self_attn_v_proj/std":0.36328127440966346,"train/train/tensor_act_model_layers_15/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn/max_abs":2.671875,"train/train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/norm":0.007701774585192801,"train/train/tensor_param_model_layers_22_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_22/act/max_abs":17.75,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_14/param/mean":0.001676467205172582,"train/train/tensor_act_model_layers_53_self_attn_q_proj/max_abs":5.3125,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/norm":6.6875,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/mean":-1.1204974725842476e-07,"train/train/layer__model_layers_40/param/mean":0.0012323198006045242,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn/norm":225.5613130784624,"train/train/tensor_act_model_layers_43/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn/mean":0.003864288330078125,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/mean":-5.759298801422119e-06,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_input_layernorm/max_abs":6,"train/train/tensor_act_model_layers_74_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn/max_abs":1.2421875,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_76/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_8/param/mean":0.0014063556927042707,"train/train/tensor_act_model_layers_62_mlp/norm":658.949244658078,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn/norm":364.0287033922374,"train/train/tensor_act_model_layers_49_self_attn_o_proj/norm":514.2676470185705,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/mean":-5.628680810332298e-08,"train/train/layer__model_layers_48/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_63/act/mean":-0.014367910531850962,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/std":8.3095138877547e-05,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/mean":0.000560760498046875,"train/train/layer_model_layers_40/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/std":0.0299072265625,"train/train/tensor_act_model_layers_85_mlp/std":0.2695312690043788,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/max_abs":0.1298828125,"train/train/tensor_act_model_layers_30_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/norm":0.003937358959638968,"train/train/tensor_act_model_layers_1_self_attn_v_proj/norm":1927.079011486621,"train/train/tensor_act_model_layers_81_post_attention_layernorm/max_abs":5.53125,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/norm":0.04938298631615474,"train/train/layer_model_layers_13/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/mean":-5.166511982679367e-07,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn/max_abs":2.71875,"train/train/tensor_act_model_layers_75_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/max_abs":0.0009002685546875,"train/train/tensor_act_model_layers_25_mlp_up_proj/std":0.2500005662434841,"train/train/layer__model_layers_2/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp/std":0.07751495894422442,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_29/param/std":0.05002479589228569,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/std":2.8722466224747594e-05,"train/train/layer_model_layers_38/act/frac_near_user_limit":0,"train/train/layer_model_layers_26/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_mlp_down_proj/mean":-0.00606536865234375,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/mean":6.218033377081156e-08,"train/train/tensor_grad_model_norm_weight/std":0.0009379540836329679,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/max_abs":0.1689453125,"train/train/tensor_grad_model_embed_tokens_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn_v_proj/mean":-0.0001734727993607521,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/mean":0.0003681182861328125,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/mean":0.00016689300537109375,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_82/grad/norm":0.07421359343088918,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/max_abs":8.296966552734375e-05,"train/train/tensor_act_model_layers_82_mlp/std":0.22534224736425928,"train/train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_52/act/norm":13635.528980728399,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/mean":-0.0001926422119140625,"train/train/tensor_act_model_layers_46/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/max_abs":0.0003604888916015625,"train/train/tensor_act_model_layers_42_self_attn/mean":-0.002109527587890625,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/mean":-3.087916411459446e-08,"train/train/tensor_act_model_layers_45/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn/norm":728.9427564260864,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/max_abs":0.0027618408203125,"train/train/layer__model_layers_0/param/std":0.05478733909749186,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/max_abs":0.10546875,"train/train/layer__model_layers_32/param/std":0.049424806719722186,"train/train/tensor_act_model_layers_37_mlp_down_proj/mean":0.009033203125,"train/train/tensor_act_model_layers_33_self_attn_q_proj/std":0.9765632553097665,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/mean":-1.3053417205810547e-05,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_input_layernorm/std":1.0000000161817295,"train/train/tensor_act_model_layers_9_self_attn_q_proj/max_abs":11.625,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/mean":7.014023140072823e-08,"train/train/tensor_act_model_layers_22_self_attn_q_proj/std":1.0156254108134688,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/max_abs":0.00128173828125,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_mlp/max_abs":2.578125,"train/train/tensor_act_model_layers_31_self_attn_v_proj/std":0.31689569746461915,"train/train/tensor_act_model_layers_45_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_q_proj/max_abs":5.6875,"train/train/tensor_param_model_layers_71_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_22/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_o_proj/mean":-0.002490997314453125,"train/train/tensor_act_model_layers_27_self_attn/std":0.0842296578061084,"train/train/tensor_act_model_layers_24_input_layernorm/std":1.0000000204890966,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/max_abs":0.00060272216796875,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/mean":-0.00012159347534179688,"train/train/layer_model_layers_38/act/mean":0.004527458777794471,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_50/grad/std":4.7443961953962714e-05,"train/train/tensor_act_model_layers_67_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/mean":0.0004425048828125,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/std":0.0244140625,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/max_abs":0.00171661376953125,"train/train/tensor_act_model_layers_58/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn/max_abs":1.578125,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/std":0.033447265625,"train/train/tensor_act_model_layers_42_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28/mean":0.0565185546875,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/max_abs":0.1201171875,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/std":0.0322265625,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/max_abs":0.002044677734375,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/max_abs":0.203125,"train/train/tensor_act_model_layers_59_self_attn_k_proj/max_abs":4.9375,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/std":6.0408562656986976e-05,"train/train/layer_model_layers_51/grad/std":5.01408033631079e-05,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/mean":4.851244739256799e-09,"train/train/tensor_act_model_layers_40_mlp/mean":-0.00670623779296875,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/mean":-1.0378425940871239e-07,"train/train/tensor_param_model_layers_17_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/mean":-2.277083694934845e-07,"train/train/tensor_act_model_layers_76_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/max_abs":0.00014781951904296875,"train/train/tensor_act_model_layers_26_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/std":0.07385304414240479,"train/train/tensor_act_model_layers_83_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/norm":0.01733677801266729,"train/train/tensor_act_model_layers_51_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_down_proj/mean":0.0070953369140625,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/max_abs":0.205078125,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_up_proj/norm":4880.735061940129,"train/train/layer_model_layers_70/grad/std":9.198851842299248e-05,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/std":0.03857421875,"train/train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_q_proj/norm":8906.533668010225,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_input_layernorm_weight/max_abs":1,"train/train/layer__model_layers_93/param/max_abs":1,"train/train/tensor_param_model_layers_14_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_28/param/mean":0.001618431436476209,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/max_abs":0.001220703125,"train/train/layer_model_layers_93/act/mean":0.025646503155048076,"train/train/tensor_act_model_layers_7_mlp_down_proj/mean":-0.002056121826171875,"train/train/tensor_act_model_layers_59_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_18/act/max_abs":17.875,"train/train/tensor_param_model_layers_22_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_38/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_post_attention_layernorm/mean":0.028778076171875,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/norm":4.09375,"train/train/tensor_act_model_layers_43_input_layernorm/norm":5792.613281252006,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_embed_tokens_weight/max_abs":0.5,"train/train/tensor_act_model_layers_12_self_attn_k_proj/max_abs":4.5625,"train/train/layer_model_layers_58/grad/std":5.651223400608228e-05,"train/train/tensor_act_model_layers_5_mlp_down_proj/norm":328.5961973752767,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/mean":-3.614695742726326e-07,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/norm":0.0005901448865833744,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp/norm":935.1437962209762,"train/train/tensor_act_model_layers_33_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_mlp/max_abs":1.03125,"train/train/tensor_act_model_layers_84_input_layernorm/norm":5792.610717775664,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/std":8.866366022200249e-05,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/mean":1.3146200217306614e-07,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/std":0.0269775390625,"train/train/tensor_act_model_layers_20_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/norm":6.75,"train/train/layer_model_layers_7/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_mlp_up_proj/mean":-0.0535888671875,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/norm":6.71875,"train/train/layer_model_layers_72/grad/std":7.310144715606646e-05,"train/train/tensor_act_model_layers_55_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn/std":0.05084947545090212,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_47_post_attention_layernorm/mean":0.0657958984375,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/max_abs":0.00023174285888671875,"train/train/tensor_param_model_layers_79_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/std":0.02490234375,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/std":0.02197265625,"train/train/layer__model_layers_79/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/max_abs":0.0013885498046875,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/norm":5.53125,"train/train/tensor_act_model_layers_87_self_attn/std":0.2739313952912764,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_28_mlp_down_proj/max_abs":0.87890625,"train/train/tensor_act_model_layers_9_self_attn_k_proj/mean":0.018829345703125,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_9/max_abs":18.5,"train/train/tensor_act_model_layers_8_post_attention_layernorm/mean":-0.00644683837890625,"train/train/layer_model_layers_69/act/mean":-0.017584910759559043,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/std":2.722065733071287e-05,"train/train/tensor_act_model_layers_35_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_77/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/std":7.97430250359019e-06,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/std":4.0071713288402765e-05,"train/train/tensor_act_model_layers_30_input_layernorm/mean":0.06036376953125,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/norm":4.09375,"train/train/layer_model_layers_30/act/mean":-0.0063915252685546875,"train/train/layer__model_layers_27/param/norm":20.306134820002672,"train/train/tensor_act_model_layers_51_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_41/act/max_abs":16.875,"train/train/tensor_act_model_layers_91_self_attn_v_proj/mean":-0.00420379638671875,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/mean":-2.0721927285194397e-06,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/mean":-9.19681042432785e-09,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/mean":5.6461431086063385e-08,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/mean":-8.440110832452774e-08,"train/train/layer_model_layers_15/act/std":0.7042407010277486,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/mean":-0.000213623046875,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/mean":2.8833746910095215e-06,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_input_layernorm/mean":0.018310546875,"train/train/tensor_act_model_layers_73_self_attn_q_proj/std":0.8212958226736491,"train/train/layer__model_layers_36/param/max_abs":1,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/norm":6.4375,"train/train/tensor_act_model_layers_13_mlp/std":0.04632613765483977,"train/train/tensor_act_model_layers_7_self_attn_k_proj/max_abs":6.9375,"train/train/tensor_param_model_layers_34_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/norm":7.0625,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_56_post_attention_layernorm/max_abs":5.8125,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/mean":0.00051116943359375,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_k_proj/std":0.9755874362662929,"train/train/tensor_act_model_layers_46_mlp_up_proj/mean":-0.09912109375,"train/train/layer__model_layers_62/param/mean":0.0015350853597132167,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/mean":-1.5709083527326584e-06,"train/train/tensor_act_model_layers_38_self_attn_k_proj/max_abs":4.59375,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn/std":0.13672054807936632,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/std":3.762084272093617e-05,"train/train/tensor_act_model_layers_21_mlp/norm":235.36975924923593,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_mlp/std":0.44531688353405463,"train/train/tensor_act_model_layers_52/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_mlp_down_proj/frac_near_user_limit":0,"train/train/layer__model_layers_92/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_post_attention_layernorm/max_abs":5.96875,"train/train/tensor_act_model_layers_45_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_o_proj/max_abs":1.078125,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/norm":0.0006098924340487422,"train/train/tensor_act_model_layers_1_self_attn_o_proj/max_abs":0.6640625,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/max_abs":0.000370025634765625,"train/train/tensor_act_model_layers_16/max_abs":18,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_down_proj/norm":232.44342835376997,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/std":0.049072265625,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/max_abs":0.2021484375,"train/train/tensor_act_model_layers_75_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_q_proj/std":1.0156250311778139,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/std":2.762277617491719e-05,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_13/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_55/act/norm":14624.971620106624,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_43_self_attn_v_proj/norm":2467.374152224016,"train/train/tensor_act_model_layers_51_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/std":0.0240478515625,"train/train/layer__model_layers_21/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/mean":-1.932494342327118e-08,"train/train/tensor_act_model_layers_20/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp/norm":518.2545092485765,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/max_abs":0.000919342041015625,"train/train/tensor_act_model_layers_65_post_attention_layernorm/max_abs":5.75,"train/train/tensor_act_model_layers_32_self_attn_q_proj/std":0.942384883898104,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/mean":7.788185030221939e-08,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_input_layernorm/mean":0.0833740234375,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/max_abs":0.228515625,"train/train/tensor_act_model_layers_69_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_post_attention_layernorm/std":1.0000000242143867,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/std":0.0341796875,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/mean":-2.100132405757904e-06,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/norm":0.006968390097326648,"train/train/layer_model_layers_5/act/max_abs":18.75,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/norm":5.78125,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/std":0.049560546875,"train/train/tensor_act_model_layers_82_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/std":2.3720054289074616e-05,"train/train/tensor_act_model_layers_84_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49/max_abs":15.625,"train/train/tensor_act_model_layers_27_post_attention_layernorm/std":1.0000000447034827,"train/train/tensor_act_model_layers_53_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_45_self_attn/norm":585.174286628416,"train/train/tensor_act_model_layers_17/mean":0.02166748046875,"train/train/tensor_param_model_layers_48_input_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_82/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_input_layernorm/max_abs":5.6875,"train/train/tensor_param_model_layers_10_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/mean":-2.710730768740177e-07,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/norm":0.0016408786265424402,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_27/grad/std":4.393491438426553e-05,"train/train/tensor_act_model_layers_4_self_attn/norm":285.5546347633865,"train/train/layer__model_layers_71/param/norm":22.533769146619147,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/max_abs":0.1220703125,"train/train/tensor_act_model_layers_67_mlp/std":0.13647701721080444,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_input_layernorm/std":1.0000001098960578,"train/train/tensor_act_model_layers_92_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp_down_proj/norm":272.5804017256328,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/norm":0.019443984584441642,"train/train/tensor_act_model_layers_73_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_26_self_attn_k_proj/std":0.8320325993585167,"train/train/tensor_act_model_layers_26_mlp_down_proj/max_abs":0.7578125,"train/train/tensor_act_model_layers_20_mlp_up_proj/norm":2349.9399528937156,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/mean":-0.00014591217041015625,"train/train/tensor_param_model_layers_57_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_66/param/std":0.056789170260322486,"train/train/tensor_act_model_layers_36_self_attn_k_proj/mean":0.014068603515625,"train/train/tensor_act_model_layers_93_self_attn_v_proj/max_abs":3.0625,"train/train/tensor_act_model_layers_53_mlp/mean":0.002960205078125,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/max_abs":0.26171875,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/norm":0.04364163056064288,"train/train/tensor_act_model_layers_56_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_48/act/max_abs":15.75,"train/train/layer_model_layers_1/act/frac_near_dtype_limit":0,"train/train/layer__model_layers_88/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/mean":1.154094934463501e-05,"train/train/tensor_act_model_layers_23_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn_q_proj/max_abs":5.28125,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/mean":2.5480985641479492e-05,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/norm":0.010724312909665341,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/mean":-0.0004482269287109375,"train/train/tensor_act_model_layers_41_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp_down_proj/norm":8393.684975032415,"train/train/tensor_act_model_layers_77_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp_down_proj/std":1.4473013879885055,"train/train/layer__model_layers_54/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/mean":-1.3140961527824402e-06,"train/train/tensor_act_model_layers_56_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_12/param/norm":19.173479456090515,"train/train/layer_model_layers_9/grad/mean":-2.9023992429471423e-07,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/mean":0.0001773834228515625,"train/train/tensor_act_model_layers_43_mlp_up_proj/mean":-0.092529296875,"train/train/layer_model_layers_87/grad/norm":0.09179509635480482,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/mean":0.0002307891845703125,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_input_layernorm/norm":5792.610351566729,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/mean":1.5615660231560469e-07,"train/train/layer_model_layers_62/grad/mean":-4.5282036856444503e-07,"train/train/tensor_act_model_layers_48_input_layernorm/mean":0.0648193359375,"train/train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/mean":0.00013446807861328125,"train/train/layer_model_layers_8/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn/std":0.2172879935130244,"train/train/tensor_act_model_layers_4_post_attention_layernorm/norm":5792.615722660921,"train/train/tensor_act_model_layers_53_self_attn_o_proj/std":0.08350553103267842,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/max_abs":0.1279296875,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/mean":-2.5134067982435226e-07,"train/train/tensor_act_model_layers_70_mlp/std":0.15820483139221242,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/mean":7.997732609510422e-08,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/norm":0.0015031746688823073,"train/train/layer_model_layers_4/act/std":0.7343688427335057,"train/train/layer__model_layers_46/param/max_abs":1,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_20/mean":0.0362548828125,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/norm":0.006683924402054046,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/mean":-8.381903171539307e-08,"train/train/tensor_act_model_layers_73_mlp_up_proj/std":0.49951309099502983,"train/train/layer_model_layers_52/act/frac_near_user_limit":0,"train/train/layer_model_layers_60/act/std":0.7046230316945215,"train/train/tensor_act_model_layers_75/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn_q_proj/std":1.0546875423855244,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/max_abs":0.0008544921875,"train/train/tensor_act_model_layers_63_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/max_abs":0.00110626220703125,"train/train/tensor_act_model_layers_45/std":1.3105629466587863,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/std":0.0267333984375,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_58/act/std":0.6619474864664003,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/std":2.5483954854091137e-05,"train/train/tensor_act_model_layers_78_mlp/std":0.18310620879977402,"train/train/tensor_act_model_layers_58_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp_up_proj/max_abs":3.375,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/norm":0.001114041827797199,"train/train/tensor_act_model_layers_89_mlp_up_proj/max_abs":4.9375,"train/train/tensor_act_model_layers_61_mlp/mean":0.013885498046875,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/max_abs":0.0005035400390625,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/max_abs":0.0003986358642578125,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/std":2.228950861943012e-05,"train/train/tensor_act_model_layers_22_mlp_down_proj/mean":0.001880645751953125,"train/train/tensor_act_model_layers_84_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_56/act/std":0.6678346946097212,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_47_self_attn_q_proj/mean":0.0102996826171875,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/mean":-0.0001544952392578125,"train/train/tensor_act_model_layers_93/mean":0.194091796875,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/norm":5.78125,"train/train/tensor_act_model_layers_27/max_abs":17.75,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/max_abs":0.000751495361328125,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/norm":0.012545137567757925,"train/train/tensor_act_model_layers_52_mlp/std":0.09143091322389986,"train/train/tensor_act_model_layers_25_self_attn/std":0.05763428783327777,"train/train/tensor_act_model_layers_37_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_65/grad/norm":0.057163645450440684,"train/train/tensor_act_model_layers_54_mlp/norm":583.6085677580277,"train/train/layer_model_layers_19/grad/std":2.9834157254397715e-05,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp/mean":0.037109375,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_norm_weight/max_abs":1,"train/train/tensor_act_model_layers_88_mlp/norm":2199.4847827219273,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/max_abs":0.2353515625,"train/train/tensor_act_model_layers_91_self_attn/std":0.24194785508573846,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/std":0.043701171875,"train/train/tensor_act_model_layers_92_input_layernorm/max_abs":5.0625,"train/train/tensor_act_model_layers_38_mlp_up_proj/std":0.3012707183171508,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_v_proj/norm":2803.560114433505,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_k_proj/std":0.8046875212667056,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/mean":-1.519918441772461e-05,"train/train/tensor_act_model_layers_39_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/std":9.502210531444461e-05,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/std":0.037109375,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/norm":5.40625,"train/train/tensor_act_model_layers_75_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_k_proj/norm":5716.382524715763,"train/train/layer_model_layers_28/grad/max_abs":0.0015106201171875,"train/train/tensor_act_model_layers_72_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer__model_layers_46/param/std":0.05228064651841919,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/max_abs":9.1552734375e-05,"train/train/tensor_act_model_layers_87_mlp_up_proj/mean":-0.156494140625,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn/mean":-0.0006465911865234375,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/max_abs":0.001068115234375,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/std":0.037841796875,"train/train/tensor_act_model_layers_51_self_attn_q_proj/max_abs":6.625,"train/train/tensor_act_model_layers_71_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/mean":-8.898787200450897e-07,"train/train/layer_model_layers_64/act/mean":-0.00724645761343149,"train/train/layer_model_layers_27/grad/mean":-5.761704839149987e-07,"train/train/tensor_act_model_layers_13_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_self_attn_v_proj/max_abs":2.234375,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/norm":4.9375,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/mean":-0.0003795623779296875,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_post_attention_layernorm/norm":5792.607788089955,"train/train/tensor_act_model_layers_15_mlp_down_proj/mean":0.00499725341796875,"train/train/tensor_act_model_layers_28_input_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_15/param/mean":0.0017190299428383386,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_post_attention_layernorm/std":1.000000081956383,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/norm":8.25,"train/train/tensor_act_model_layers_13_mlp_up_proj/max_abs":2.328125,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/norm":0.013173735429509783,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_83/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_12/act/mean":-0.01602473625769982,"train/train/tensor_act_model_layers_41_self_attn_k_proj/std":0.8408221144482447,"train/train/tensor_act_model_layers_47_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn_k_proj/max_abs":4.46875,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn_v_proj/std":0.48242220519727763,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_post_attention_layernorm/norm":5792.608032229695,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51/max_abs":15.375,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/mean":-0.000255584716796875,"train/train/tensor_act_model_layers_87_mlp_down_proj/mean":0.016998291015625,"train/train/tensor_act_model_layers_93_self_attn_o_proj/std":0.23952149475257986,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/max_abs":0.0004596710205078125,"train/train/tensor_act_model_layers_58_self_attn_q_proj/max_abs":5.5,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_q_proj/max_abs":6.25,"train/train/tensor_act_model_layers_73_self_attn_q_proj/max_abs":6.09375,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/norm":0.011659512206522483,"train/train/tensor_act_model_layers_10_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/mean":1.317821443080902e-07,"train/train/tensor_act_model_layers_0_post_attention_layernorm/norm":5792.37500000497,"train/train/tensor_act_model_layers_35_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/std":0.037109375,"train/train/tensor_act_model_layers_44_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/norm":0.027512777242862482,"train/train/tensor_act_model_layers_78/norm":9496.191283530137,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_o_proj/norm":130.5136547005455,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/norm":0.014074063844699145,"train/train/tensor_act_model_layers_60_self_attn_q_proj/max_abs":7.34375,"train/train/tensor_act_model_layers_19_self_attn_o_proj/mean":-0.0003561973571777344,"train/train/tensor_act_model_layers_17_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/std":0.033447265625,"train/train/tensor_act_model_layers_20_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/norm":5.09375,"train/train/tensor_act_model_layers_71_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_embed_tokens_weight/std":0.1123046875,"train/train/tensor_act_model_layers_40_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73/std":1.5078249273628064,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/norm":3.59375,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/mean":-1.5506520867347717e-06,"train/train/tensor_act_model_layers_82_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_q_proj/norm":5503.609130998851,"train/train/tensor_act_model_layers_23_self_attn_q_proj/std":1.0703132448402652,"train/train/tensor_act_model_layers_45_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn/std":0.09692538284413871,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/std":3.4413552223912286e-05,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/norm":7.59375,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/norm":4.65625,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_34/act/std":0.6616645679130654,"train/train/tensor_act_model_layers_56_self_attn_o_proj/mean":0.00016689300537109375,"train/train/tensor_act_model_layers_56_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/std":2.074117554329835e-05,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/mean":0.0004673004150390625,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/max_abs":0.12451171875,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/mean":-0.000743865966796875,"train/train/layer_model_layers_5/grad/max_abs":0.0020294189453125,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/max_abs":0.1240234375,"train/train/tensor_act_model_layers_37_mlp/max_abs":0.5390625,"train/train/tensor_act_model_layers_36_self_attn_v_proj/frac_near_user_limit":0,"train/train/global/param/norm":225.71885762059597,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/max_abs":0.2373046875,"train/train/tensor_param_model_layers_72_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_72_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/mean":-3.419816493988037e-06,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/std":0.00015355499195381206,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/std":0.045166015625,"train/train/tensor_act_model_layers_85_post_attention_layernorm/mean":0.0714111328125,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/norm":0.002620050053370902,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp_down_proj/mean":0.006988525390625,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/norm":0.0007566574870738182,"train/train/tensor_act_model_layers_86_self_attn_k_proj/norm":6304.8888225799965,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_down_proj/norm":2591.8679217349736,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/mean":1.4870602171868086e-07,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_mlp_up_proj/mean":-0.097412109375,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_77/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_input_layernorm_weight/mean":1,"train/train/layer_model_layers_59/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/max_abs":0.0001506805419921875,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74/norm":8901.133750682293,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/norm":2.859375,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/std":0.050048828125,"train/train/layer__model_layers_42/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_input_layernorm/mean":0.0819091796875,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/norm":3.96875,"train/train/tensor_act_model_layers_7_mlp/std":0.0593261722429299,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/std":0.046875,"train/train/tensor_act_model_layers_56/norm":7645.3067751952685,"train/train/layer__model_layers_15/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/std":5.430362784034171e-05,"train/train/tensor_act_model_layers_9_mlp_down_proj/std":0.04656994492114231,"train/train/tensor_act_model_layers_51_mlp/mean":-0.003265380859375,"train/train/tensor_act_model_layers_17_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/norm":5.46875,"train/train/layer_model_layers_68/grad/mean":2.21489139726493e-07,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/std":1.5330439994922957e-05,"train/train/tensor_act_model_layers_65_self_attn_k_proj/mean":0.01727294921875,"train/train/tensor_act_model_layers_53_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_input_layernorm/std":0.9990266947166466,"train/train/tensor_act_model_layers_16_self_attn_q_proj/mean":-0.05645751953125,"train/train/tensor_act_model_layers_42_post_attention_layernorm/mean":0.06304931640625,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/std":0.0228271484375,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/mean":0.0002155303955078125,"train/train/layer_model_layers_54/act/max_abs":15.125,"train/train/layer__model_layers_63/param/std":0.05594365891263966,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/mean":-0.000274658203125,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/norm":7.90625,"train/train/tensor_act_model_layers_72_input_layernorm/std":1.0000019539128526,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/max_abs":0.002593994140625,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/norm":4.875,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/mean":-0.000591278076171875,"train/train/tensor_act_model_layers_82_self_attn_o_proj/mean":-0.000946044921875,"train/train/tensor_param_model_layers_75_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_45_self_attn_o_proj/max_abs":1.046875,"train/train/tensor_act_model_layers_38_self_attn_k_proj/norm":5631.528684464766,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/max_abs":0.2333984375,"train/train/tensor_act_model_layers_47_mlp_up_proj/mean":-0.0936279296875,"train/train/tensor_act_model_layers_65_mlp_down_proj/norm":737.1329428876805,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/max_abs":0.00077056884765625,"train/train/tensor_act_model_layers_74_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/mean":-9.778887033462524e-09,"train/train/tensor_act_model_layers_24_self_attn/std":0.07727164999452271,"train/train/layer__model_layers_50/param/max_abs":1,"train/train/tensor_act_model_layers_56_mlp_up_proj/std":0.39453160880799554,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/norm":3.703125,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/norm":0.01848015119287437,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/max_abs":0.220703125,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/max_abs":0.00015544891357421875,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/norm":0.002316259552393731,"train/train/layer_model_layers_16/act/max_abs":18,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/norm":4.90625,"train/train/tensor_act_model_layers_18_mlp_down_proj/mean":0.00493621826171875,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/norm":0.013175856120996398,"train/train/tensor_act_model_layers_65_input_layernorm/max_abs":5.75,"train/train/layer__model_layers_0/param/norm":22.208891536527663,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_93_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/mean":-1.2195669114589691e-06,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/norm":0.0004974314405614553,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/std":0.00010926984418128254,"train/train/tensor_param_model_layers_63_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/norm":5.03125,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/norm":2.953125,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/mean":-0.00015735626220703125,"train/train/tensor_param_model_layers_82_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_49/std":1.2949382268871472,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn/norm":410.7656186242378,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/mean":-3.273598849773407e-07,"train/train/tensor_act_model_layers_43_post_attention_layernorm/norm":5792.609130860697,"train/train/layer_model_layers_0/grad/std":0.0005562691912214341,"train/train/layer__model_layers_85/param/norm":23.633141784896058,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/max_abs":0.0015716552734375,"train/train/tensor_param_model_layers_43_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_62_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/std":4.882509180542497e-05,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/mean":7.231719791889191e-07,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_57_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_90/act/max_abs":12.9375,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/max_abs":0.126953125,"train/train/tensor_act_model_layers_46_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn_q_proj/std":0.9218805397228964,"train/train/tensor_param_model_layers_24_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/norm":5.03125,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/norm":0.014910296546649751,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/norm":0.007071341682051809,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/mean":4.234607331454754e-09,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/mean":-0.000110626220703125,"train/train/tensor_act_model_layers_62_mlp_down_proj/max_abs":1.0234375,"train/train/tensor_act_model_layers_54_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/max_abs":0.000942230224609375,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/std":5.5796068811770944e-05,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/norm":5.375,"train/train/tensor_act_model_layers_55_post_attention_layernorm/max_abs":5.75,"train/train/tensor_act_model_layers_21_self_attn_v_proj/max_abs":1.7890625,"train/train/layer_model_layers_9/grad/max_abs":0.002960205078125,"train/train/tensor_act_model_layers_91_mlp_down_proj/mean":-0.009796142578125,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/norm":6.34375,"train/train/tensor_act_model_layers_0_self_attn_q_proj/norm":2746.9560793204273,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/std":0.025146484375,"train/train/tensor_act_model_layers_82_self_attn/std":0.11097682848320566,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_q_proj/mean":-0.028533935546875,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/max_abs":0.00015544891357421875,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn/norm":791.9043731140305,"train/train/tensor_act_model_layers_87_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/norm":0.006261571959775733,"train/train/tensor_act_model_layers_48_mlp_up_proj/norm":3950.450947147074,"train/train/tensor_act_model_layers_80_post_attention_layernorm/std":1.0000021904682965,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_61_self_attn_v_proj/std":0.41699357747154353,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/max_abs":0.000308990478515625,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/max_abs":0.208984375,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/norm":5.375,"train/train/tensor_act_model_layers_83_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/std":0.0308837890625,"train/train/tensor_act_model_layers_9_mlp/mean":-0.0001195669174194336,"train/train/tensor_act_model_layers_25_mlp_down_proj/std":0.048828440158542166,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/mean":-1.3851968105882406e-07,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/norm":0.022355703877557812,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/mean":-7.855123840272427e-07,"train/train/layer_model_layers_22/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_25/grad/max_abs":0.0020751953125,"train/train/tensor_act_model_layers_16_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/std":0.0517578125,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/mean":-6.973277777433395e-08,"train/train/tensor_act_model_layers_51_self_attn_o_proj/max_abs":0.94140625,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/mean":0.000732421875,"train/train/tensor_act_model_layers_22_mlp_up_proj/norm":2564.638824551998,"train/train/tensor_param_model_layers_10_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/mean":2.7474015951156616e-08,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/std":3.74005018417913e-05,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/mean":9.988434612751007e-08,"train/train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/norm":6.65625,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/mean":-2.4199485778808594e-05,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/norm":0.0028365653836514176,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_post_attention_layernorm/max_abs":5.71875,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14/max_abs":18.125,"train/train/tensor_act_model_layers_89_mlp_down_proj/max_abs":4.71875,"train/train/tensor_act_model_layers_31_self_attn_o_proj/max_abs":0.98046875,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/std":0.037109375,"train/train/tensor_act_model_layers_39_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn/std":0.151857836758432,"train/train/tensor_act_model_layers_68_self_attn_v_proj/norm":2688.050375344441,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/std":0.0400390625,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/max_abs":0.000396728515625,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/max_abs":0.00048828125,"train/train/tensor_act_model_layers_73_self_attn_o_proj/max_abs":0.953125,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/std":0.0244140625,"train/train/tensor_act_model_layers_58_self_attn_o_proj/max_abs":1.328125,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/norm":9.4375,"train/train/tensor_act_model_layers_71_mlp/max_abs":1.4453125,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_60/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/norm":0.03934188115996687,"train/train/tensor_act_model_layers_40_self_attn/std":0.09253101271404147,"train/train/layer__model_layers_51/param/norm":21.586128358276756,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/max_abs":0.173828125,"train/train/tensor_act_model_layers_1_self_attn_o_proj/std":0.09350619070153761,"train/train/tensor_grad_model_norm_weight/max_abs":0.005859375,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/max_abs":0.00074005126953125,"train/train/tensor_act_model_layers_25_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/max_abs":0.197265625,"train/train/tensor_act_model_layers_60_self_attn_o_proj/max_abs":1.53125,"train/train/layer_model_layers_27/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_input_layernorm_weight/mean":1,"train/train/layer_model_layers_49/grad/std":5.105408098834952e-05,"train/train/tensor_act_model_layers_33_mlp_up_proj/std":0.279296875,"train/train/tensor_act_model_layers_13_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/mean":-5.116453394293785e-08,"train/train/tensor_act_model_layers_51_self_attn/max_abs":0.94140625,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/max_abs":0.00031280517578125,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/std":2.933161678538038e-05,"train/train/tensor_act_model_layers_12_mlp/mean":0.003559112548828125,"train/train/tensor_act_model_layers_13_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/std":0.043212890625,"train/train/tensor_act_model_layers_35/std":1.3105630205640844,"train/train/tensor_act_model_layers_2_mlp_down_proj/norm":605.1734946376591,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/norm":3.984375,"train/train/tensor_act_model_layers_47/std":1.3085996883883886,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/mean":-3.767490852624178e-08,"train/train/tensor_act_model_layers_45_self_attn_q_proj/mean":0.06640625,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_q_proj/mean":0.1328125,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/mean":0.0001010894775390625,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/mean":-4.819594323635101e-08,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_k_proj/norm":6642.859365658513,"train/train/tensor_act_model_layers_9_mlp/std":0.04656994492114231,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/max_abs":7.486343383789062e-05,"train/train/layer_model_layers_32/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/max_abs":0.12890625,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/mean":6.62636011838913e-07,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/mean":-6.341934204101562e-05,"train/train/tensor_act_model_layers_42_self_attn/std":0.06774943701729534,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/max_abs":0.1484375,"train/train/tensor_act_model_layers_69_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/norm":0.005069363812417906,"train/train/tensor_act_model_layers_21/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_73_mlp_up_proj/max_abs":3.765625,"train/train/tensor_act_model_layers_1_mlp_up_proj/max_abs":2.734375,"train/train/tensor_act_model_layers_59_mlp_up_proj/std":0.4160159652780207,"train/train/layer_model_layers_30/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp/norm":1070.645826143877,"train/train/tensor_act_model_layers_20/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/std":0.025634765625,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/norm":7.5,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/mean":2.5617191568017006e-07,"train/train/tensor_act_model_layers_47_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/max_abs":6.866455078125e-05,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/mean":-0.0003643035888671875,"train/train/tensor_act_model_layers_24_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_15/grad/norm":0.034212912675467014,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/mean":5.7872384786605835e-06,"train/train/tensor_act_model_layers_73_mlp_down_proj/norm":955.9582227644396,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/max_abs":0.19921875,"train/train/tensor_act_model_layers_38_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93/norm":20422.747997067072,"train/train/tensor_act_model_layers_27_self_attn_q_proj/std":0.9677751172548684,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/norm":0.0004884721338166856,"train/train/tensor_act_model_layers_12_mlp_up_proj/std":0.2087408322331305,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/max_abs":0.000518798828125,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/max_abs":0.000946044921875,"train/train/tensor_act_model/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn_k_proj/norm":7867.877006882439,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/norm":6.78125,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/mean":-2.1507730707526207e-08,"train/train/layer__model_layers_7/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/std":4.356926431525307e-05,"train/train/tensor_act_model_layers_31/max_abs":17.75,"train/train/tensor_act_model_layers_54_input_layernorm/norm":5792.609375000587,"train/train/tensor_act_model_layers_32_self_attn_q_proj/max_abs":5.1875,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/mean":-0.0003204345703125,"train/train/tensor_act_model_layers_54_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_input_layernorm/norm":5792.609130862642,"train/train/tensor_act_model_layers_7_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/norm":7.1875,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/norm":0.020329380357289907,"train/train/layer__model_layers_45/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_84/param/norm":23.850261234156324,"train/train/tensor_act_model_layers_44_self_attn_k_proj/max_abs":4.8125,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/mean":-0.00041961669921875,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_56/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/std":5.864069115187902e-05,"train/train/layer_model_layers_88/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_o_proj/mean":-0.0013980865478515625,"train/train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/mean":-0.000732421875,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/std":2.9778539124633644e-05,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/norm":7.1875,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/max_abs":0.00057220458984375,"train/train/tensor_act_model_layers_49_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn_v_proj/std":0.33203125515103554,"train_steps_per_second":0.403,"train/train/tensor_act_model_layers_88_self_attn_o_proj/max_abs":3.828125,"train/train/tensor_act_model_layers_45_post_attention_layernorm/std":0.9960937799191938,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_46/act/max_abs":16.125,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/max_abs":0.00075531005859375,"train/train/tensor_act_model_layers_13_self_attn_v_proj/std":0.2558594136960946,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/std":4.446017014707961e-05,"train/train/tensor_act_model_layers_54_mlp_up_proj/std":0.3916028372346674,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/norm":0.0027105403830525693,"train/train/tensor_act_model_layers_6_mlp_up_proj/norm":2446.231050470047,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/std":6.605057450653743e-05,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/std":0.034912109375,"train/train/tensor_act_model_layers_54_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_42_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_input_layernorm/norm":5792.605712891289,"train/train/tensor_act_model_layers_62/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/std":0.0001012809158904555,"train/train/layer_model_layers_67/grad/std":7.257972259080665e-05,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/max_abs":0.390625,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/max_abs":0.00079345703125,"train/train/tensor_act_model_layers_54/norm":7654.155023952801,"train/train/tensor_act_model_layers_66_mlp/std":0.12548926367913205,"train/train/tensor_act_model_layers_72_self_attn_q_proj/std":0.8574244883677773,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/norm":6.75,"train/train/layer__model_layers_81/param/norm":23.954342964794506,"train/train/tensor_act_model_layers_57_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/mean":1.8812716007232666e-07,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/norm":0.0290244469653202,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/std":0.00011910199319384159,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/max_abs":0.12890625,"train/train/tensor_act_model_layers_26_self_attn_o_proj/std":0.048523078549410646,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/norm":0.0008022957710689732,"train/train/tensor_act_model_layers_46_self_attn_q_proj/norm":6056.136213618214,"train/train/tensor_act_model_layers_55_mlp/mean":-0.005828857421875,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/std":0.0255126953125,"train/train/tensor_act_model_layers_57_self_attn_q_proj/mean":0.01251220703125,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_self_attn_k_proj/mean":0.005359649658203125,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/norm":0.0007693356380936601,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/mean":-0.0001068115234375,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/max_abs":0.00023937225341796875,"train/train/tensor_act_model_layers_18_self_attn_v_proj/std":0.28125007188645207,"train/train/layer__model_layers_68/param/std":0.05679480734805322,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/mean":0.0004787445068359375,"train/train/tensor_act_model_layers_3_mlp/std":0.08569370817184105,"train/train/tensor_act_model_layers_6_mlp/norm":319.4020248436044,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/std":0.0242919921875,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_mlp_up_proj/mean":-0.07568359375,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/mean":0.0001277923583984375,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_input_layernorm/mean":0.0469970703125,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/mean":3.7066638469696045e-07,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/max_abs":0.1748046875,"train/train/tensor_act_model_layers_40_input_layernorm/norm":5792.604370118252,"train/train/tensor_act_model_layers_63_mlp_down_proj/norm":661.8512047819681,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_58_mlp/std":0.09643587703889751,"train/train/tensor_act_model_layers_48_mlp/norm":480.44002816296006,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/mean":4.3422915041446686e-08,"train/train/tensor_act_model_layers_44_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/mean":0.00014400482177734375,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/mean":6.532669067382812e-05,"train/train/tensor_act_model_layers_19_self_attn_k_proj/mean":-0.048828125,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/max_abs":0.1533203125,"train/train/tensor_act_model_layers_54/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/mean":-0.001007080078125,"train/train/tensor_act_model_layers_39_mlp/mean":-0.007293701171875,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp/std":0.047119144060759836,"train/train/layer_model_layers_19/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/max_abs":0.19921875,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/std":6.020604625366477e-05,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_o_proj/max_abs":1.03125,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn/max_abs":1.2890625,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_input_layernorm/max_abs":6,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/std":0.0263671875,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/max_abs":0.3125,"train/train/layer_model_layers_67/act/std":0.7157979228894702,"train/train/tensor_act_model_layers_25_input_layernorm/max_abs":6.0625,"train/train/tensor_act_model_layers_73/max_abs":13.125,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/norm":4.25,"train/train/tensor_act_model_layers_26_self_attn_o_proj/max_abs":0.70703125,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/max_abs":5.28125,"train/train/tensor_act_model_layers_21/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn/max_abs":1.6484375,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/mean":0.000217437744140625,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_input_layernorm/norm":5792.605957034985,"train/train/tensor_act_model_layers_77_mlp_down_proj/norm":1003.4255223609749,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/std":0.020751953125,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_32/act/norm":13738.818527804407,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/max_abs":0.134765625,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/max_abs":0.000621795654296875,"train/train/tensor_act_model_layers_70_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/max_abs":0.0004367828369140625,"train/train/tensor_param_model_layers_74_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/std":9.829060890182983e-05,"train/train/tensor_act_model_layers_56/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_v_proj/mean":0.003444671630859375,"train/train/tensor_act_model_layers_68_mlp/mean":-0.00455474853515625,"train/train/tensor_act_model_layers_89_post_attention_layernorm/norm":5792.612060552601,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/norm":0.012144385763028792,"train/train/tensor_act_model_layers_64_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/std":8.022866249011446e-05,"train/train/layer__model_layers_79/param/std":0.05847010804242444,"train/train/tensor_act_model_layers_70_self_attn_q_proj/max_abs":7.8125,"train/train/layer_model_layers_5/grad/mean":-2.49507819562453e-07,"train/train/tensor_act_model_layers_62_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/mean":7.438939064741135e-08,"train/train/tensor_act_model_layers_86_self_attn_v_proj/mean":0.0055694580078125,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/std":0.00017713902186538414,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/norm":0.00040555834271054666,"train/train/layer_model_layers_84/grad/max_abs":0.0010986328125,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/max_abs":0.000606536865234375,"train/train/tensor_param_model_layers_52_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/std":0.023681640625,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_87/param/frac_near_user_limit":0,"train/train/layer_model_layers_19/grad/mean":-5.303798519458488e-07,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/max_abs":0.2451171875,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/mean":1.9173603504896164e-07,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/std":0.0240478515625,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/mean":1.589651219546795e-07,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_26/param/mean":0.0018307452268793876,"train/train/tensor_act_model_layers_87_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/std":2.6109328251245958e-05,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_87/param/std":0.05953358582640135,"train/train/tensor_act_model_layers_17_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_input_layernorm/mean":0.05523681640625,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/max_abs":0.1962890625,"train/train/tensor_act_model_layers_66_input_layernorm/norm":5792.604125979581,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/max_abs":0.00159454345703125,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/mean":-0.000614166259765625,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/mean":0.001068115234375,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/std":4.8529611926618415e-05,"train/train/tensor_act_model_layers_2_post_attention_layernorm/mean":0.02801513671875,"train/train/tensor_act_model_layers_29_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_88/grad/max_abs":0.0017547607421875,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/norm":0.024830058781749757,"train/train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn/mean":-0.0012378692626953125,"train/train/tensor_act_model_layers_7_input_layernorm/mean":-0.00501251220703125,"train/train/tensor_act_model_layers_69_mlp_down_proj/mean":-0.012786865234375,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/mean":6.866455078125e-05,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn/mean":-0.008880615234375,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/mean":-0.0003223419189453125,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/mean":-4.1433395381318405e-08,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/std":7.524178376999864e-05,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/std":2.3576377838344043e-05,"train/train/tensor_act_model_layers_78_mlp_down_proj/norm":1065.8288129052837,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/mean":5.832407623529434e-08,"train/train/tensor_act_model_layers_24_mlp_down_proj/norm":281.0360299481865,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/norm":4.75,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/max_abs":0.000614166259765625,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/std":0.00018933224629339205,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/mean":-6.034970283508301e-06,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/max_abs":0.2373046875,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/norm":6.40625,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/mean":3.6030542105436325e-08,"train/train/layer_model_layers_47/act/mean":-0.004001323993389423,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn_o_proj/max_abs":0.57421875,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/norm":4.375,"train/train/tensor_act_model_layers_34_mlp_up_proj/max_abs":2.53125,"train/train/layer__model_layers_23/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn_v_proj/max_abs":1.703125,"train/train/tensor_act_model_layers_25_mlp/std":0.048828440158542166,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/mean":-1.2842938303947449e-06,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/mean":-6.12996518611908e-06,"train/train/tensor_act_model_layers_90_input_layernorm/max_abs":5.59375,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/std":0.05615234375,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/mean":1.255422830581665e-06,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/std":0.044189453125,"train/train/tensor_act_model_layers_68_mlp_up_proj/norm":4947.021487019485,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp_down_proj/max_abs":0.8671875,"train/train/tensor_act_model_layers_16/mean":0.019012451171875,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/mean":0.000392913818359375,"train/train/tensor_act_model_layers_81_self_attn_q_proj/std":1.1738330758445796,"train/train/tensor_act_model_layers_36_mlp/max_abs":0.71875,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp_down_proj/norm":448.9106466684905,"train/train/tensor_act_model_layers_82_self_attn_k_proj/mean":0.03662109375,"train/train/tensor_act_model_layers_32_mlp/max_abs":0.427734375,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_q_proj/max_abs":6.90625,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn/mean":0.0002644062042236328,"train/train/layer_model_layers_8/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/std":1.6354549542296563e-05,"train/train/tensor_act_model_layers_65_self_attn_o_proj/mean":0.0012133121490478516,"train/train/tensor_act_model_layers_92_self_attn_q_proj/std":0.9834000467112608,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/max_abs":0.000274658203125,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/std":0.00013729821003109644,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/std":4.22765691767882e-05,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_norm/std":1.0000003725289606,"train/train/tensor_act_model_layers_73_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/std":0.00011651855689571345,"train/train/layer_model_layers_24/act/norm":14462.113181650186,"train/train/layer_model_layers_73/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/std":0.0001810956260394819,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/norm":0.009130622738306471,"train/train/layer__model_layers_65/param/norm":22.633350496280926,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/norm":0.034673126266889016,"train/train/tensor_act_model_layers_43_mlp/max_abs":1.09375,"train/train/layer__model_layers_74/param/norm":22.842296465384123,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/std":0.0252685546875,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/max_abs":0.000553131103515625,"train/train/tensor_act_model_layers_89_input_layernorm/std":0.9960953656351786,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/max_abs":0.265625,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/mean":4.773028194904327e-09,"train/train/tensor_act_model_layers_84/mean":0.12939453125,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_92/act/max_abs":17.75,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/std":2.0102106615160353e-05,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/std":0.04443359375,"train/train/tensor_act_model_layers_53_mlp_down_proj/max_abs":0.8828125,"train/train/layer__model_layers_64/param/std":0.05647512471582737,"train/train/tensor_act_model_layers_22_self_attn_v_proj/mean":-0.004608154296875,"train/train/tensor_act_model_layers_34_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_o_proj/norm":636.1683186661928,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/max_abs":0.1806640625,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/max_abs":0.002777099609375,"train/train/tensor_act_model_layers_40_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_89/grad/norm":0.09924163852377801,"train/train/tensor_param_model_layers_16_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/norm":0.0205396677292968,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/mean":4.246830940246582e-07,"train/train/tensor_act_model_layers_85_input_layernorm/max_abs":5.21875,"train/train/tensor_act_model_layers_55_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_62_self_attn/max_abs":1.7109375,"train/train/tensor_act_model_layers_40_input_layernorm/std":0.9960941314696534,"train/train/tensor_act_model_layers_15_self_attn/std":0.0456547808414322,"train/train/tensor_param_model_layers_21_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/std":6.402769373518303e-05,"train/train/tensor_param_model_layers_9_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_83/act/mean":0.004166823167067308,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_47_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_18/grad/norm":0.03197380404638714,"train/train/tensor_act_model_layers_31/norm":7598.582595507411,"train/train/tensor_act_model_layers_80_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/max_abs":0.166015625,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/max_abs":0.111328125,"train/train/tensor_act_model_layers_18_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_66/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn/mean":-0.002857208251953125,"train/train/tensor_act_model_layers_6_self_attn_v_proj/mean":-0.00330352783203125,"train/train/tensor_act_model_layers_87_self_attn_o_proj/std":0.2739313952912764,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_post_attention_layernorm/std":1.0000025238809922,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_self_attn_o_proj/norm":525.9160700821067,"train/train/tensor_act_model_layers_45_input_layernorm/mean":0.068603515625,"train/train/tensor_act_model_layers_71_input_layernorm/std":1.000001566017683,"train/train/tensor_act_model_layers_87_self_attn/mean":-0.007049560546875,"train/train/tensor_act_model_layers_75_mlp/max_abs":1.453125,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/mean":-3.1990930438041687e-07,"train/train/tensor_act_model_layers_3_self_attn_v_proj/norm":2122.912405903216,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/max_abs":0.00133514404296875,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/max_abs":0.2197265625,"train/train/tensor_act_model_layers_50_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/norm":0.0011438113876516313,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/max_abs":0.0002880096435546875,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/std":8.613419596509173e-05,"train/train/tensor_act_model_layers_2_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/max_abs":0.10986328125,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/mean":-0.000644683837890625,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/max_abs":0.0020294189453125,"train/train/tensor_param_model_layers_0_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_47/max_abs":15.9375,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/mean":0.0003490447998046875,"train/train/tensor_act_model_layers_27_self_attn_q_proj/max_abs":4.9375,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/norm":0.043663136216974416,"train/train/tensor_act_model_layers_46_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_input_layernorm/mean":0.0865478515625,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/norm":0.028201346618715435,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/norm":4.03125,"train/train/tensor_act_model_layers_72_self_attn/std":0.07739773695129841,"train/train/tensor_param_model_layers_77_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/max_abs":0.232421875,"train/train/tensor_act_model_layers_26_self_attn_o_proj/norm":281.24286589023365,"train/train/tensor_grad_model_norm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_v_proj/norm":2444.6227798057243,"train/train/tensor_act_model_layers_35_input_layernorm/std":0.9960938322777808,"train/train/tensor_act_model_layers_26_self_attn_q_proj/max_abs":5.3125,"train/train/tensor_act_model_layers_55_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/std":0.00010613672229087415,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/mean":3.725290298461914e-09,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_input_layernorm/std":1.0000001844018527,"train/train/tensor_act_model_layers_15_self_attn_k_proj/max_abs":5.65625,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_30/grad/std":4.881860105342266e-05,"train/train/tensor_act_model_layers_61_input_layernorm/mean":0.0474853515625,"train/train/tensor_act_model_layers_29_post_attention_layernorm/mean":0.05035400390625,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/max_abs":0.01361083984375,"train/train/layer_model_layers_79/grad/max_abs":0.00106048583984375,"train/train/layer_model_layers_61/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp_down_proj/max_abs":8.6875,"train/train/tensor_act_model_layers_48_mlp_up_proj/std":0.37988444036218333,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/norm":0.012396636911466169,"train/train/tensor_act_model_layers_29_mlp/mean":0.01116943359375,"train/train/tensor_act_model_layers_70_self_attn_o_proj/max_abs":5.53125,"train/train/tensor_act_model_layers_2_mlp/max_abs":0.80859375,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_v_proj/mean":0.002376556396484375,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/max_abs":0.279296875,"train/train/tensor_act_model_layers_20_mlp/std":0.03747578474855653,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/std":6.083918511316061e-05,"train/train/tensor_act_model_layers_77/mean":0.06732177734375,"train/train/layer__model_layers_64/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp_down_proj/max_abs":0.9140625,"train/train/tensor_act_model_layers_28_mlp/max_abs":0.87890625,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/max_abs":0.255859375,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/max_abs":0.1435546875,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_k_proj/max_abs":5.5,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/norm":0.023193760918604655,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_post_attention_layernorm/norm":5792.605957038299,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/max_abs":0.000579833984375,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_mlp/max_abs":1.2265625,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/max_abs":0.1630859375,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_18/param/norm":19.702734680907877,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/mean":-2.736342139542103e-07,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/norm":3.890625,"train/train/tensor_act_model_layers_46_self_attn/std":0.10180817654077728,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/norm":0.0009756158174057284,"train/train/tensor_act_model_layers_54_mlp_down_proj/std":0.09985381874840575,"train/train/tensor_act_model_layers_3_mlp/max_abs":0.671875,"train/train/layer_model_layers_0/act/mean":0.0036107576810396635,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn/norm":536.6007892014339,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/mean":-1.0879011824727058e-07,"train/train/tensor_act_model_layers_13_self_attn/mean":-0.0007410049438476562,"train/train/tensor_param_model_layers_83_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/max_abs":0.24609375,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_88_self_attn_o_proj/std":0.38964977718304616,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/norm":0.014878845136601574,"train/train/layer__model_layers_69/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_o_proj/norm":406.01484164935357,"train/train/tensor_act_model_layers_63_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_o_proj/norm":1587.689742085104,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/std":1.7726801771705627e-05,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_q_proj/std":0.9804727071705026,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/max_abs":0.1201171875,"train/train/layer_model_layers_19/act/norm":13468.92383253543,"train/train/tensor_param_model_layers_15_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp/mean":0.00386810302734375,"train/train/tensor_act_model_layers_36_self_attn_o_proj/std":0.08606055346647698,"train/train/tensor_param_model_layers_9_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_74_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn_o_proj/norm":310.7889711898425,"train/train/tensor_act_model_layers_82_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/mean":-5.0067901611328125e-05,"train/train/tensor_param_model_layers_7_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_11/param/norm":19.604767885086652,"train/train/layer__model_layers_90/param/norm":25.513113203066773,"train/train/tensor_act_model_layers_17_self_attn_o_proj/std":0.04382387288809894,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/mean":4.98257577419281e-08,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/norm":7.0625,"train/train/tensor_param_model_layers_13_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp/mean":0.016754150390625,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/norm":0.018163574376264313,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/std":0.00012976021380724315,"train/train/tensor_param_model_layers_25_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_37/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/mean":4.38690185546875e-05,"train/train/layer_model_layers_45/act/mean":0.007022454188420222,"train/train/tensor_act_model_layers_11_input_layernorm/mean":-0.00545501708984375,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/std":0.041748046875,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/norm":0.0032824666458394205,"train/train/tensor_act_model_layers_88_mlp_up_proj/std":0.7216824956072735,"train/train/tensor_act_model_layers_19_mlp/std":0.03790302245849343,"train/train/tensor_act_model_layers_76_post_attention_layernorm/mean":0.048828125,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/mean":-6.580352783203125e-05,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn/max_abs":0.91015625,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/std":0.032470703125,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/norm":4.96875,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/norm":5.4375,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_/max_abs":2.124464511871338,"train/train/tensor_act_model_layers_21_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/std":0.0303955078125,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp/mean":-0.00647735595703125,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/std":7.993216348674113e-05,"train/train/tensor_param_model_layers_31_input_layernorm_weight/mean":1,"train/train/layer__model_layers_39/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/max_abs":0.000843048095703125,"train/train/tensor_act_model_layers_84_self_attn_q_proj/std":1.0683648285386618,"train/train/layer_model_layers_46/act/std":0.689825798353411,"train/train/tensor_act_model_layers_28_self_attn_k_proj/max_abs":5.4375,"train/train/tensor_act_model_layers_14_self_attn/std":0.05133229969242981,"train/train/tensor_act_model_layers_61_self_attn_k_proj/mean":0.0496826171875,"train/train/tensor_act_model_layers_44_input_layernorm/std":0.9990249644964188,"train/train/tensor_act_model_layers_64_self_attn_v_proj/std":0.48730576993825864,"train/train/layer__model_layers_83/param/max_abs":1,"train/train/tensor_act_model_layers_50_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_86/param/std":0.05852219792405747,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/norm":6.125,"train/train/tensor_act_model_layers_34_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_k_proj/norm":5753.575366421397,"train/train/tensor_act_model_layers_47/mean":0.0777587890625,"train/train/tensor_act_model_layers_28_self_attn/max_abs":0.8125,"train/train/tensor_act_model_layers_91_self_attn_v_proj/max_abs":3.53125,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/max_abs":7.2479248046875e-05,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/norm":12.25,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/mean":2.3171305656433105e-06,"train/train/tensor_act_model_layers_56_mlp_up_proj/norm":4085.672940089366,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/std":3.691831383098449e-05,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/norm":7.15625,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/mean":4.363059997558594e-05,"train/train/tensor_act_model_layers_92_post_attention_layernorm/norm":5792.612060547055,"train/train/tensor_act_model_layers_89_mlp/mean":-0.042724609375,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/norm":0.006601520713132168,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_30/grad/max_abs":0.0011138916015625,"train/train/tensor_act_model_layers_81_post_attention_layernorm/std":1.0000024046719964,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn_o_proj/norm":589.9938815628393,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/mean":-1.0817311704158783e-06,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/mean":0.00010061264038085938,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/max_abs":0.2236328125,"train/train/tensor_act_model_layers_73_input_layernorm/norm":5792.61108398471,"train/train/tensor_act_model_layers_46_mlp_up_proj/max_abs":3.0625,"train/train/tensor_act_model_layers_66_self_attn_q_proj/std":1.007813570110729,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/norm":5.09375,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/max_abs":0.00014972686767578125,"train/train/layer_model_layers_44/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn/std":0.16162248252325284,"train/train/layer_model_layers_69/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp/norm":343.9711612161347,"train/train/tensor_act_model_layers_49_mlp_up_proj/max_abs":2.796875,"train/train/tensor_act_model_layers_55_self_attn_o_proj/max_abs":1.6015625,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/max_abs":0.0001392364501953125,"train/train/tensor_act_model_layers_68_mlp_down_proj/std":0.139161025220266,"train/train/tensor_act_model_layers_50_self_attn/norm":327.12346802635545,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/mean":-0.0014801025390625,"train/train/tensor_act_model_layers_50_self_attn_v_proj/norm":2043.2781136003468,"train/train/tensor_act_model_layers_10_mlp_down_proj/max_abs":0.5859375,"train/train/tensor_act_model_layers_75/max_abs":12.75,"train/train/layer_model_layers_89/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/max_abs":0.0001983642578125,"train/train/tensor_act_model_layers_66_input_layernorm/max_abs":5.8125,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/norm":0.00256876173602623,"train/train/tensor_act_model_layers_58_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn/max_abs":1.2890625,"train/train/layer__model_layers_10/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn/frac_near_dtype_limit":0,"total_flos":3.6392499412992e+16,"train/train/tensor_param_model_layers_35_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_18_mlp/norm":213.3784408100035,"train/train/tensor_act_model_layers_35_mlp_up_proj/std":0.29296885172524273,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/mean":1.8699211068451405e-08,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_42/act/max_abs":16.75,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_38/act/norm":14276.320729783138,"train/train/tensor_act_model_layers_93_self_attn_k_proj/std":0.8535225625705506,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/max_abs":0.0002040863037109375,"train/train/tensor_act_model_layers_27_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/mean":-0.00647735595703125,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/mean":0.0004444122314453125,"train/train/tensor_act_model_layers_25_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/std":1.0000000558793527,"train/train/layer_model_layers_51/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_o_proj/std":0.03894152385645313,"train/train/tensor_act_model_layers_93_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/mean":1.330590748693794e-08,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/mean":0.000186920166015625,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/mean":-1.1515803635120392e-06,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/max_abs":0.0003814697265625,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/norm":0.010636284459634672,"train/train/tensor_act_model_layers_10_mlp_down_proj/mean":0.003143310546875,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/norm":4.5625,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/std":6.146091308731745e-05,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/std":0.00011135830977075626,"train/train/tensor_act_model_layers_71_self_attn_v_proj/std":0.3637705790376561,"train/train/tensor_act_model_layers_18_self_attn/std":0.06152458013198962,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/max_abs":0.0003719329833984375,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69/max_abs":13.25,"train/train/tensor_act_model_layers_93_self_attn_v_proj/norm":2938.3299095150846,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_44/act/max_abs":16.5,"train/train/tensor_act_model_layers_68_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_post_attention_layernorm/norm":5792.61206054898,"train/train/tensor_act_model_layers_46/norm":7614.537744950414,"train/train/tensor_param_model_layers_78_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/mean":4.772096872329712e-06,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn/max_abs":1.4296875,"train/train/tensor_act_model_layers_78_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/mean":-1.7369166016578674e-07,"train/train/tensor_act_model_layers_58_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_27/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp_up_proj/mean":-0.066162109375,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/std":2.7263912833485558e-05,"train/train/tensor_param_model_layers_11_input_layernorm_weight/std":0,"train/train/layer_model_layers_12/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_o_proj/mean":-0.0012378692626953125,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/norm":6.53125,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/norm":6.21875,"train/train/tensor_act_model_layers_28_mlp_up_proj/std":0.2792969817047982,"train/train/tensor_act_model_layers_65_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/max_abs":0.12451171875,"train/train/tensor_act_model_layers_64_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/mean":2.477318048477173e-07,"train/train/tensor_act_model_layers_16_self_attn_o_proj/std":0.04046695038088668,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_37/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/std":4.835738406560576e-05,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/std":0.02734375,"train/train/layer_model_layers_29/act/norm":13819.934824010006,"train/train/tensor_act_model_layers_17_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/std":0.0255126953125,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/norm":0.00461701305821524,"train/train/layer_model_layers_26/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/norm":5.375,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_13/param/max_abs":1,"train/train/tensor_act_model_layers_7_mlp_up_proj/std":0.2343751509983848,"train/train/tensor_act_model_layers_17_self_attn_k_proj/mean":-0.0657958984375,"train/train/tensor_act_model_layers_66_self_attn_v_proj/norm":2692.5864298343354,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/norm":6.3125,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/mean":5.279434844851494e-08,"train/train/tensor_act_model_layers_24_self_attn_k_proj/max_abs":4.71875,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/norm":0.0010405054440290413,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/std":2.791201966011172e-05,"train/train/tensor_act_model_layers_32_mlp/norm":264.5384415227059,"train/train/layer__model_layers_40/param/norm":21.15683275600639,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/max_abs":0.0004749298095703125,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/std":3.031005450139641e-05,"train/train/tensor_act_model_layers_89_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/norm":0.031112547420988193,"train/train/tensor_param_model_layers_73_input_layernorm_weight/mean":1,"train/train/layer_model_layers_38/grad/max_abs":0.002105712890625,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/norm":6.5625,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/norm":0.024752521477965344,"train/train/tensor_act_model_layers_14_self_attn_k_proj/norm":5664.03818572037,"train/train/layer__model_layers_42/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/std":8.488696429714655e-05,"train/train/tensor_act_model_layers_48_self_attn_k_proj/mean":0.0133209228515625,"train/train/tensor_act_model_layers_2_mlp_down_proj/max_abs":0.80859375,"train/train/layer__model_layers_4/param/std":0.0475034907081187,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/mean":-4.7907233238220215e-06,"train/train/layer_model_layers_72/grad/max_abs":0.0016021728515625,"train/train/layer_model_layers_38/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp/norm":186.95777352632697,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/mean":-2.6426278054714203e-08,"train/train/tensor_act_model_layers_92_mlp_up_proj/std":0.8632829329530356,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_29/param/norm":20.27477265593871,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/std":0.03955078125,"train/train/tensor_act_model_layers_70_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_post_attention_layernorm/std":0.9960939519545406,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/std":6.986093785748307e-05,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/max_abs":0.2373046875,"train/train/tensor_act_model_layers_57_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_43/act/max_abs":16.5,"train/train/layer__model_layers_59/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/std":0.036865234375,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/max_abs":0.1396484375,"train/train/layer__model_layers_12/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_78/grad/norm":0.07336093643956493,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_80/act/max_abs":11.6875,"train/train/tensor_act_model_layers_20_input_layernorm/norm":5792.605468753015,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn_q_proj/std":1.1250000579489587,"train/train/tensor_act_model_layers_44_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_self_attn/std":0.17359298616151067,"train/train/tensor_act_model_layers_86_mlp/norm":1429.1296013645515,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/norm":5.96875,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/std":3.6044955529498295e-05,"train/train/tensor_act_model_layers_31_input_layernorm/max_abs":6.125,"train/train/tensor_act_model_layers_64_input_layernorm/max_abs":5.84375,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/mean":2.1364539861679077e-06,"train/train/tensor_act_model_layers_40_mlp_up_proj/mean":-0.0855712890625,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/mean":-0.000415802001953125,"train/train/layer_model_layers_10/act/frac_near_dtype_limit":0,"train/train/layer__model_layers_85/param/std":0.058378779708442075,"train/train/global/act/mean":-0.06364485442572262,"train/train/tensor_act_model_layers_10_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_post_attention_layernorm/max_abs":6.0625,"train/train/tensor_act_model_layers_71_self_attn_q_proj/mean":0.0135955810546875,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/max_abs":0.13671875,"train/train/tensor_act_model_layers_42_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/mean":-0.00018310546875,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/max_abs":0.0002994537353515625,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/norm":0.029897197676749792,"train/train/tensor_act_model_layers_75_mlp_down_proj/norm":899.3549603284915,"train/train/tensor_act_model_layers_84_self_attn_v_proj/mean":-0.0079345703125,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/norm":6.53125,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/std":0.022705078125,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/max_abs":0.1279296875,"train/train/tensor_act_model_layers_10_mlp/std":0.04400715317961084,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_69_input_layernorm/norm":5792.6107177741615,"train/train/tensor_act_model_layers_25_self_attn_v_proj/norm":1767.4753256873094,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_q_proj/mean":-0.015899658203125,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/mean":0.000537872314453125,"train/train/tensor_act_model_layers_61_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/norm":0.009573038841702132,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_88/grad/mean":-2.1895106777246582e-07,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/norm":5.375,"train/train/tensor_act_model_layers_64/frac_near_user_limit":0,"train/train/layer_model_layers_84/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/std":2.203517955341895e-05,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/std":6.014771407516233e-05,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/max_abs":0.11083984375,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/max_abs":0.2099609375,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/std":5.3544711159244274e-05,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn/std":0.095337186747636,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/mean":-5.675246939063072e-07,"train/train/tensor_act_model_layers_91_mlp/std":0.6093875359505121,"train/train/tensor_act_model_layers_69_self_attn_v_proj/norm":2346.988332797199,"train/train/tensor_act_model_layers_51_self_attn_v_proj/std":0.3833017753175127,"train/train/tensor_act_model_layers_49_input_layernorm/max_abs":5.71875,"train/train/layer__model_layers_92/param/std":0.06374941248432099,"train/train/tensor_act_model_layers_20_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn/norm":1001.9470209422302,"train/train/tensor_act_model_layers_55_self_attn_k_proj/mean":0.0034942626953125,"train/train/tensor_act_model_layers_79_self_attn_q_proj/mean":-0.088623046875,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/mean":7.343292236328125e-05,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/mean":-0.000354766845703125,"train/train/layer_model_layers_16/grad/mean":-3.542716203725282e-07,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/std":6.72486477392842e-05,"train/train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/norm":0.01785995469775333,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/max_abs":0.000240325927734375,"train/train/tensor_act_model_layers_63_self_attn/mean":-0.000804901123046875,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/std":0.0274658203125,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/mean":-0.0003814697265625,"train/train/layer_model_layers_62/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_q_proj/std":0.9248063108597135,"train/train/tensor_act_model_layers_27_mlp_down_proj/std":0.04321289803348628,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/norm":0.0454117969503669,"train/train/tensor_param_model_layers_60_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/std":0.0269775390625,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/max_abs":0.173828125,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/mean":-0.0001678466796875,"train/train/tensor_act_model_layers_77_self_attn_v_proj/max_abs":3.8125,"train/train/tensor_param_model_layers_38_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/norm":0.027378400917957516,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/std":9.523085200911355e-05,"train/train/tensor_act_model_layers_56_post_attention_layernorm/norm":5792.612426762334,"train/train/tensor_act_model_layers_69_mlp/max_abs":1.4765625,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/max_abs":0.002838134765625,"train/train/layer_model_layers_21/grad/norm":0.04038451935460296,"train/train/tensor_param_model_layers_31_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/mean":0.00041961669921875,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/max_abs":0.000904083251953125,"train/train/tensor_act_model_layers_59_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/mean":0.0002899169921875,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/max_abs":0.1259765625,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/max_abs":0.248046875,"train/train/tensor_act_model_layers_5_input_layernorm/max_abs":5.09375,"train/train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/norm":0.010850662125433367,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/max_abs":0.232421875,"train/train/tensor_act_model_layers_31_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/std":5.384217974495197e-05,"train/train/tensor_act_model_layers_40_mlp_down_proj/norm":390.37375003261207,"train/train/tensor_act_model_layers_22_input_layernorm/norm":5792.601806644468,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/mean":1.6898848116397858e-06,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/mean":1.0423536878079176e-07,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/mean":-1.6540288925170898e-06,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/std":2.9219312284376742e-05,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/std":5.890393869722108e-06,"train/train/tensor_param_model_layers_67_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/mean":6.914138793945312e-05,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/std":4.10152111149215e-05,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/std":2.614068083414992e-05,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/norm":7.6875,"train/train/tensor_act_model_layers_55_self_attn_q_proj/max_abs":6.09375,"train/train/tensor_param_model_layers_46_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp_down_proj/std":0.15502991563568388,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/mean":-7.799826562404633e-07,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/norm":0.017702931159794048,"train/train/tensor_act_model_layers_77_input_layernorm/std":1.0000025704470024,"train/train/layer__model_layers_27/param/max_abs":1,"train/train/tensor_param_model_layers_54_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_75_post_attention_layernorm/std":1.0000025704470024,"train/train/tensor_act_model_layers_33_self_attn_v_proj/max_abs":1.8046875,"train/train/tensor_act_model_layers_84_self_attn/norm":936.057846637477,"train/train/tensor_act_model_layers_11_self_attn_q_proj/mean":-0.0445556640625,"train/train/tensor_act_model_layers_54_self_attn_q_proj/std":0.8750044341485886,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/std":0.00011270128836249562,"train/train/tensor_param_model_layers_25_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/norm":0.0015911357916912083,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/norm":0.014009064239213886,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/mean":-0.00031280517578125,"train/train/tensor_act_model_layers_62_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/std":0.0007448444459582899,"train/train/tensor_act_model_layers_86_mlp_down_proj/std":0.24609377412568836,"train/train/layer_model_layers_42/act/norm":13986.202212679878,"train/train/tensor_act_model_layers_39_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_v_proj/max_abs":2.828125,"train/train/tensor_act_model_layers_12_mlp_down_proj/norm":234.43811557411908,"train/train/tensor_act_model_layers_57_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/mean":0.00012302398681640625,"train/train/tensor_act_model_layers_56_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_61/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/norm":0.006636468203408594,"train/train/tensor_param_model_layers_88_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/norm":0.001364563655089555,"train/train/tensor_act_model_layers_71_mlp_up_proj/std":0.5000001639127463,"train/train/tensor_act_model_layers_89/max_abs":12.3125,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/max_abs":0.0009613037109375,"train/train/layer__model_layers_58/param/mean":0.0013510574602671607,"train/train/layer_model_layers_43/act/norm":14193.951144887717,"train/train/tensor_act_model_layers_62_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55/frac_near_dtype_limit":0,"train/train/layer_model_layers_58/grad/norm":0.04578523569825172,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/max_abs":0.20703125,"train/train/tensor_act_model_layers_17_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/norm":4.15625,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/std":3.191193733993487e-05,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/std":3.257229061990639e-05,"train/train/tensor_act_model_layers_66_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/mean":-2.651941031217575e-07,"train/train/tensor_act_model_layers_30_self_attn_q_proj/norm":6176.1973286529155,"train/train/tensor_act_model_layers_65_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_input_layernorm/mean":0.0367431640625,"train/train/tensor_act_model_layers_7_input_layernorm/std":1.0000000149884727,"train/train/layer_model_layers_77/grad/std":8.909165589267315e-05,"train/train/tensor_act_model_layers_54/max_abs":15.125,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/std":3.057908892615728e-05,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/max_abs":0.234375,"train/train/tensor_act_model_layers_21_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/norm":3.203125,"train/train/tensor_act_model_layers_69_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_mlp_up_proj/std":0.26953133292818865,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/max_abs":0.13671875,"train/train/tensor_act_model_layers_3_mlp_up_proj/mean":-0.0675048828125,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/norm":0.022950679651389933,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/std":0.0283203125,"train/train/tensor_act_model_layers_44_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/std":5.289541819924839e-05,"train/train/tensor_act_model_layers_53_self_attn_v_proj/mean":-0.003780364990234375,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/norm":6.625,"train/train/layer_model_layers_8/act/std":0.7842670281018099,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/norm":5.6875,"train/train/tensor_act_model_layers_19_mlp_up_proj/mean":-0.052734375,"train/train/tensor_act_model_layers_28_self_attn_q_proj/norm":5998.032445296902,"train/train/tensor_act_model_layers_34_self_attn/norm":502.29755068105896,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/std":0.00013865031632160127,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/max_abs":0.1455078125,"train/train/layer_model_layers_2/grad/mean":-2.4016934930813294e-07,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/norm":5.40625,"train/train/tensor_act_model_layers_79_post_attention_layernorm/norm":5792.610839844933,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/mean":1.8690479919314384e-07,"train/train/tensor_act_model_layers_79_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/max_abs":0.00055694580078125,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_36/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_post_attention_layernorm/std":0.9960938322777808,"train/train/tensor_act_model_layers_45_mlp_down_proj/std":0.07971219672767066,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/mean":0.00025177001953125,"train/train/layer_model_layers_41/act/mean":-0.005915128267728365,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/std":0.040283203125,"train/train/tensor_act_model_layers_61_self_attn/std":0.09619530942408828,"train/train/tensor_act_model_layers_26_self_attn_q_proj/norm":5338.651785481668,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/mean":-2.0329025574028492e-08,"train/train/tensor_param_model_layers_63_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_53_self_attn/std":0.08350553103267842,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/max_abs":0.000576019287109375,"train/train/tensor_act_model_layers_49_self_attn/std":0.08874831462803683,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/max_abs":0.00069427490234375,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/std":0.037109375,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_mlp_down_proj/std":0.04614258064794786,"train/train/tensor_act_model_layers_57_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/max_abs":0.000270843505859375,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/mean":-9.5367431640625e-06,"train/train/layer_model_layers_53/grad/max_abs":0.000926971435546875,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_post_attention_layernorm/mean":0.05584716796875,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/norm":5.375,"train/train/tensor_act_model_layers_35_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/std":3.9593231571594005e-05,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/norm":0.031984751376279555,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_mlp/std":0.09716858216671954,"train/train/tensor_act_model_layers_24_input_layernorm/max_abs":6.03125,"train/train/layer_model_layers_30/act/max_abs":17.75,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_down_proj/mean":0.003574371337890625,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_14_input_layernorm/std":1.000000056199495,"train/train/layer_model_layers_70/act/std":0.7580400669503349,"train/train/tensor_act_model_layers_33_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn/mean":0.0016231536865234375,"train/train/layer_model_layers_68/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60/norm":7758.403817285047,"train/train/tensor_act_model_layers_65_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_51/act/mean":-0.017717801607572116,"train/train/layer_model_layers_70/grad/norm":0.07456970585619513,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/std":6.788131057715362e-05,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/mean":0.00063323974609375,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/mean":-4.1676685214042664e-07,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/mean":-2.089887857437134e-06,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/norm":0.00047685432403184094,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn_q_proj/norm":4562.3042994834595,"train/train/tensor_param_model_layers_49_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_34_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/norm":6.34375,"train/train/tensor_act_model_layers_76_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_41/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_k_proj/std":0.7460941095001793,"train/train/tensor_param_model_layers_65_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_34/act/norm":13825.37205270746,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn_v_proj/max_abs":2.21875,"train/train/layer__model_layers_48/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_q_proj/norm":6513.542706218802,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_22/act/std":0.6887395602329687,"train/train/tensor_act_model_layers_58_mlp_up_proj/mean":-0.105712890625,"train/train/tensor_act_model_layers_43_input_layernorm/std":0.9960939519545408,"train/train/tensor_act_model_layers_6_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3/std":1.5097896433217652,"train/train/tensor_act_model_layers_33_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp_up_proj/norm":5228.603433577185,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/norm":5.6875,"train/train/layer_model_layers_3/grad/max_abs":0.0027618408203125,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/mean":9.66247171163559e-09,"train/train/layer_model_layers_85/grad/std":9.800603789931273e-05,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/max_abs":0.000270843505859375,"train/train/layer__model_layers_7/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/max_abs":0.00022602081298828125,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/std":4.668630827416948e-05,"train/train/tensor_act_model_layers_61_self_attn_v_proj/max_abs":2.296875,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/max_abs":0.1484375,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/max_abs":0.0010986328125,"train/train/tensor_act_model_layers_50_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_77/param/std":0.05789583076726329,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/mean":8.726119995117188e-05,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/norm":0.003972072647067029,"train/train/tensor_act_model_layers_28_self_attn_q_proj/mean":0.0124664306640625,"train/train/layer__model_layers_67/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_mlp_up_proj/mean":-0.162841796875,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_83/param/norm":24.256623612170348,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/std":9.66428582306501e-06,"train/train/layer_model_layers_12/act/max_abs":18.125,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/std":3.450870789348752e-05,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_mlp/std":0.12744240348204688,"train/train/layer_model_layers_63/grad/mean":-6.287923887627731e-07,"train/train/layer_model_layers_63/act/std":0.7113895457422618,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/std":0.0286865234375,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_input_layernorm/norm":5792.609130862056,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/mean":1.0952353477478027e-05,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/max_abs":0.0003261566162109375,"train/train/tensor_act_model_layers_81_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/mean":8.405186235904694e-07,"train/train/layer__model_layers_86/param/mean":0.0012528409080088789,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp_up_proj/std":0.247070644212583,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/std":8.60784576082156e-05,"train/train/tensor_act_model_layers_59_mlp/mean":0.017669677734375,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/norm":3.5,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_26_self_attn_q_proj/mean":-0.003200531005859375,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/max_abs":0.00168609619140625,"train/train/tensor_act_model_layers_78_self_attn_v_proj/norm":2631.4421417300437,"train/train/tensor_act_model_layers_33_self_attn/std":0.047546551212269104,"train/train/layer_model_layers_20/act/std":0.6384361518013436,"train/train/tensor_act_model_layers_12_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/mean":0.000423431396484375,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/norm":0.027267964107004924,"train/train/tensor_act_model_layers_90_self_attn_v_proj/norm":2957.7047705981868,"train/train/tensor_act_model_layers_15_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/max_abs":0.000797271728515625,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/max_abs":0.1083984375,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_61/grad/std":6.771797445310684e-05,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/norm":7.09375,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/mean":-0.00081634521484375,"train/train/tensor_act_model_layers_14_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/max_abs":0.2314453125,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/max_abs":0.2412109375,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/std":0.0260009765625,"train/train/tensor_act_model_layers_84_mlp/mean":0.0037994384765625,"train/train/layer_model_layers_86/grad/mean":6.78161202745393e-07,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_input_layernorm/norm":5792.608032229766,"train/train/tensor_param_model_layers_90_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/max_abs":0.00013256072998046875,"train/train/tensor_act_model_layers_65_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/std":0.02197265625,"train/train/tensor_act_model_layers_93_mlp_up_proj/std":0.9482457154008982,"train/train/tensor_param_model_layers_92_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/mean":1.0360963642597198e-07,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/norm":0.013212419714363448,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/max_abs":0.00066375732421875,"train/train/tensor_act_model_layers_58_self_attn_o_proj/mean":-0.0012683868408203125,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/norm":6,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/std":1.1553376553222227e-05,"train/train/layer_model_layers_46/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/norm":0.018166035370692436,"train/train/tensor_act_model_layers_65_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn/norm":636.5367902032095,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/max_abs":0.00140380859375,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/mean":-1.595914363861084e-05,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/std":4.911570563423547e-05,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_mlp_up_proj/mean":-0.0555419921875,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/max_abs":0.0002994537353515625,"train/train/tensor_act_model_layers_27_self_attn/max_abs":1.03125,"train/train/tensor_act_model_layers_49_mlp_down_proj/mean":-0.00557708740234375,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/mean":-8.521601557731628e-08,"train/train/tensor_act_model_layers_6_self_attn_o_proj/max_abs":1.1171875,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/norm":11.8125,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_49/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/max_abs":0.00058746337890625,"train/train/tensor_act_model_layers_53_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/norm":0.03549434427421132,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/mean":-1.3509998098015785e-07,"train/train/tensor_act_model_layers_63_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_self_attn_q_proj/std":0.964843935329404,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/std":3.950051960595954e-05,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/mean":1.4457327779382467e-07,"train/train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/norm":0.012787775630907567,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/mean":1.3469252735376358e-07,"train/train/layer__model_layers_89/param/norm":25.22982253826808,"train/train/layer_model_layers_1/act/mean":-0.017426710862379808,"train/train/tensor_act_model_layers_1_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_93/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_input_layernorm/max_abs":5.28125,"train/train/tensor_act_model_layers_69_mlp_up_proj/norm":5072.8019684816545,"train/train/tensor_act_model_layers_82_self_attn_q_proj/std":0.9267636812464419,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/std":2.535686864979841e-05,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/mean":-1.1292286217212677e-07,"train/train/layer_model_layers_23/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp_up_proj/mean":-0.10107421875,"train/train/tensor_act_model_layers_42_self_attn_k_proj/std":0.8369166103686473,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_q_proj/max_abs":5.9375,"train/train/tensor_act_model_layers_43_mlp/std":0.06652866336642194,"train/train/tensor_act_model_layers_71_self_attn_q_proj/norm":4760.598957954674,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/norm":0.013558531004623963,"train/train/tensor_act_model_layers_36_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/std":7.809355660692658e-05,"train/train/tensor_act_model_layers_36_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_83_self_attn_q_proj/std":1.1250004039869643,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/norm":5.4375,"train/train/tensor_act_model_layers_33_self_attn_v_proj/mean":0.0013780593872070312,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/std":0.0439453125,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/mean":-0.000217437744140625,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/mean":-2.580927684903145e-07,"train/train/tensor_act_model_layers_59_self_attn_q_proj/norm":4996.665163945248,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/norm":4.84375,"train/train/tensor_act_model_layers_71_post_attention_layernorm/norm":5792.61389160489,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_post_attention_layernorm/mean":0.05267333984375,"train/train/tensor_act_model_layers_55_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/max_abs":0.0010986328125,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/mean":1.3096723705530167e-07,"train/train/tensor_act_model_layers_39_self_attn_v_proj/mean":-0.0015964508056640625,"train/train/tensor_act_model_layers_80_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/max_abs":0.25390625,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp/std":0.04443359571498825,"train/train/layer__model_layers_1/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/std":0.06005859375,"train/train/tensor_act_model_layers_82_mlp_up_proj/norm":5962.8148192902345,"train/train/tensor_act_model_layers_65_mlp_up_proj/max_abs":3.90625,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/norm":0.017419523414000646,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_o_proj/max_abs":1.859375,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/norm":0.01339926736558279,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/mean":-1.171603798866272e-06,"train/train/tensor_act_model_layers_35/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_input_layernorm/mean":0.024566650390625,"train/train/tensor_act_model_layers_44_mlp_up_proj/mean":-0.088134765625,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_k_proj/norm":5140.350776210231,"train/train/tensor_param_model_layers_10_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_self_attn_q_proj/norm":6316.132689417019,"train/train/tensor_act_model_layers_76_input_layernorm/norm":5792.606079105301,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/std":8.018318911889547e-05,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/max_abs":0.16015625,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/max_abs":0.001434326171875,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_68_self_attn_k_proj/max_abs":4.9375,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/max_abs":0.19140625,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/norm":5.375,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/max_abs":0.130859375,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/max_abs":0.0010223388671875,"train/train/tensor_act_model_layers_32_self_attn/std":0.04113863411245636,"train/train/tensor_act_model_layers_7_self_attn/norm":582.3355205777782,"train/train/tensor_act_model_layers_54_self_attn_o_proj/mean":-0.00235748291015625,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/std":0.0006416866632034233,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/norm":0.03768929547133928,"train/train/tensor_act_model_layers_18_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/norm":7.09375,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn/max_abs":0.70703125,"train/train/global/act/max_abs":30,"train/train/tensor_act_model_layers_68_self_attn_o_proj/std":0.1801796702091887,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/mean":0.000560760498046875,"train/train/tensor_act_model_layers_15_self_attn_v_proj/norm":1691.4959133414764,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_input_layernorm/std":1.0000000502914177,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/std":3.805546742055697e-05,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/max_abs":0.1357421875,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/mean":-9.778887033462524e-08,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/mean":0.000537872314453125,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp/std":0.11413755383558184,"train/train/tensor_act_model_layers_70_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_input_layernorm/norm":5792.613525391998,"train/train/tensor_act_model_layers_35_self_attn_q_proj/std":0.9560565982561822,"train/train/tensor_param_model_layers_3_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/mean":0.000579833984375,"train/train/tensor_act_model_layers_23_self_attn_v_proj/mean":0.003231048583984375,"train/train/tensor_act_model_layers_74_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp_up_proj/norm":6185.890899768009,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/std":0.0260009765625,"train/train/tensor_act_model_layers_44_mlp_down_proj/norm":407.67899836444633,"train/train/tensor_act_model_layers_49_post_attention_layernorm/max_abs":5.625,"train/train/layer_model_layers_26/grad/mean":-5.937493791260325e-07,"train/train/layer_model_layers_81/grad/max_abs":0.00107574462890625,"train/train/layer__model_layers_14/param/norm":19.690674347263986,"train/train/tensor_act_model_layers_57_self_attn_o_proj/norm":551.9051980183211,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/norm":0.0032413615329395075,"train/train/tensor_act_model_layers_35_post_attention_layernorm/mean":0.07275390625,"train/train/tensor_act_model_layers_12_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/max_abs":0.0002040863037109375,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/std":0.0279541015625,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/max_abs":0.1259765625,"train/train/tensor_act_model_layers_52/std":1.3007873099704292,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/max_abs":0.162109375,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/mean":6.454065442085266e-07,"train/train/layer__model_layers_28/param/norm":20.302022538176463,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/norm":0.0006100981901690814,"train/train/tensor_act_model_layers_47_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_down_proj/std":0.6093875359505121,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/mean":-1.730397343635559e-06,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/norm":0.02182593633537442,"train/train/tensor_act_model_layers_90/mean":0.1396484375,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/mean":1.9237631931900978e-08,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_o_proj/mean":0.0008821487426757812,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/mean":-9.107589721679688e-05,"train/train/tensor_param_model_layers_20_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/std":0.03173828125,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83/std":1.8183669265278861,"train/train/tensor_act_model_layers_61_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp_up_proj/std":0.5859377543130958,"train/train/layer_model_layers_3/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_down_proj/mean":-0.0018405914306640625,"train/train/tensor_param_model_layers_86_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_33/norm":7551.016232987131,"train/train/tensor_act_model_layers_75_post_attention_layernorm/norm":5792.610595703397,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/std":0.00014404442433218792,"train/train/tensor_param_model_layers_33_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/std":0.03369140625,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/mean":-7.390975952148438e-05,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/norm":0.01999076078511025,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/std":0.051025390625,"train/train/tensor_act_model_layers_84_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/norm":0.0038091160954004943,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/norm":7.21875,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/norm":4.875,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/mean":-3.9674341678619385e-07,"train/train/layer__model_layers_72/param/max_abs":1,"train/train/tensor_act_model_layers_24_mlp_up_proj/norm":2703.6972317801856,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/norm":0.0013979395207512728,"train/train/tensor_act_model_layers_43_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_70/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/norm":0.0022974233994436893,"train/train/layer_model_layers_41/grad/max_abs":0.00139617919921875,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/norm":5.65625,"train/train/tensor_act_model_layers_27_input_layernorm/norm":5792.610473636902,"train/train/tensor_act_model_layers_93_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/std":5.448453387955198e-05,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/std":2.6366569756251687e-05,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/std":5.258528829485294e-05,"train/train/tensor_act_model_layers_17_post_attention_layernorm/std":1.0000000534346314,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/norm":0.0031975264876855283,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/max_abs":0.1591796875,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/std":0.042724609375,"train/train/tensor_act_model_layers_41_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_14/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_post_attention_layernorm/norm":5792.6118164064765,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_mlp_up_proj/norm":3147.571099078682,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_34/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91/norm":14236.642071449513,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_post_attention_layernorm/norm":5792.605712891227,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/norm":5.8125,"train/train/tensor_act_model_layers_10_mlp/mean":0.003143310546875,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/max_abs":0.134765625,"train/train/tensor_param_model_layers_54_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/norm":5.375,"train/train/tensor_param_model_layers_80_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_29_mlp/std":0.07251018947901292,"train/train/tensor_act_model_layers_56_mlp_up_proj/max_abs":2.71875,"train/train/tensor_act_model_layers_2_self_attn_k_proj/std":1.0781250621961491,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/std":1.696331298139428e-05,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_83/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/mean":-0.000125885009765625,"train/train/tensor_act_model_layers_68/mean":0.0535888671875,"train/train/tensor_act_model_layers_67_self_attn_o_proj/mean":-0.00034880638122558594,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/std":2.3322252123482137e-05,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/norm":5.71875,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/std":3.352149934135225e-05,"train/train/tensor_act_model_layers_41_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_15_input_layernorm/std":1.0000000429572529,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/mean":-1.866370439529419e-06,"train/train/tensor_act_model_layers_52_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/norm":7.6875,"train/train/tensor_act_model_layers_87/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/std":0.046142578125,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58/max_abs":14.4375,"train/train/tensor_act_model_layers_87_self_attn_q_proj/mean":0.0010957717895507812,"train/train/tensor_act_model_layers_71_self_attn/max_abs":0.9453125,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_61/act/norm":14649.944712483166,"train/train/tensor_act_model_layers_92_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp/norm":407.67899836444633,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_10/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/max_abs":0.169921875,"train/train/tensor_param_model_layers_47_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_8/grad/norm":0.039934789377304276,"train/train/layer_model_layers_88/act/norm":19923.426149853545,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/max_abs":0.00019550323486328125,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/mean":-8.344650268554688e-05,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/max_abs":0.2197265625,"train/train/tensor_act_model_layers_24_self_attn_v_proj/norm":1765.6630720467376,"train/train/tensor_act_model_layers_52_self_attn_q_proj/mean":-0.019622802734375,"_timestamp":1.786251636718838e+09,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/norm":3.453125,"train/train/tensor_act_model_layers_67_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn_k_proj/norm":5064.199003348172,"train/train/tensor_act_model_layers_33_mlp_down_proj/std":0.051330680973195014,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/mean":0.000469207763671875,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_88_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_54_post_attention_layernorm/std":1.0000001937150769,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/mean":1.0372605174779892e-07,"train/train/tensor_param_model_layers_19_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64/norm":8062.20234795416,"train/train/layer__model_layers_59/param/norm":21.775263703914128,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_o_proj/mean":-0.0012874603271484375,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/std":4.145104944020751e-05,"train/train/tensor_act_model_layers_92_self_attn_o_proj/max_abs":2.796875,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/max_abs":0.11376953125,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp/mean":0.0012454986572265625,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_25/param/frac_near_user_limit":0,"train/train/layer_model_layers_54/act/mean":-0.013184767503004808,"train/train/tensor_act_model_layers_55_mlp_up_proj/mean":-0.1016845703125,"train/train/tensor_act_model_layers_80_mlp/mean":0.00266265869140625,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/norm":0.0018707541421862466,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/std":2.6801780403609526e-05,"train/train/tensor_act_model_layers_5_mlp/norm":328.5961973752767,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/norm":3.078125,"train/train/tensor_act_model_layers_57_self_attn_k_proj/max_abs":5.5625,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/max_abs":0.000377655029296875,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/max_abs":0.00164031982421875,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/max_abs":0.216796875,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/std":9.667698280676409e-06,"train/train/tensor_param_model_layers_14_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/std":9.454258798973046e-05,"train/train/tensor_act_model_layers_76_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_q_proj/max_abs":5.84375,"train/train/tensor_act_model_layers_44_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/mean":-9.679794311523438e-05,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/max_abs":0.0001697540283203125,"train/train/tensor_param_model_layers_47_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_89_self_attn_v_proj/mean":-0.0098419189453125,"train/train/layer__model_layers_23/param/mean":0.0015742737863811427,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/max_abs":0.126953125,"train/train/layer_model_layers_48/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6/mean":-0.00507354736328125,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/max_abs":0.0020599365234375,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/std":0.0576171875,"train/train/tensor_act_model_layers_45_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_56/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_k_proj/std":0.9921915526382402,"train/train/tensor_act_model_layers_10_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_v_proj/mean":0.0011281967163085938,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn/max_abs":2.203125,"train/train/tensor_act_model_layers_47_mlp_down_proj/norm":483.783498865299,"train/train/layer_model_layers_0/grad/max_abs":0.01361083984375,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/std":6.245116226643392e-05,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/mean":-2.4903565645217896e-06,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/mean":2.689659595489502e-05,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93/std":3.5156442853610583,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_k_proj/mean":-0.01861572265625,"train/train/layer_model_layers_47/grad/mean":-5.854401220211559e-07,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/norm":0.027101396381466174,"train/train/tensor_act_model_layers_50_self_attn/mean":-0.001644134521484375,"train/train/tensor_act_model_layers_18_mlp/mean":0.00493621826171875,"train/train/tensor_param_model_layers_64_input_layernorm_weight/mean":1,"train/train/layer_model_layers_72/grad/mean":1.1341308691072483e-06,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/max_abs":0.000186920166015625,"train/train/tensor_act_model_layers_62_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34/norm":7571.335161545135,"train/train/tensor_act_model_layers_27_self_attn_v_proj/norm":1881.289667381618,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/mean":-0.0002574920654296875,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/std":2.6105100264291385e-05,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/norm":3.265625,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/max_abs":0.1337890625,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/max_abs":0.10498046875,"train/train/tensor_act_model_layers_36_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/std":4.649533952196237e-05,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/mean":-8.0108642578125e-05,"train/train/layer__model_layers_43/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/std":1.7352794284018073e-05,"train/train/tensor_act_model_layers_21_input_layernorm/max_abs":5.96875,"train/train/layer_model_layers_41/act/std":0.6608874675348593,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_q_proj/mean":-0.05596923828125,"train/train/tensor_act_model_layers_50_self_attn_v_proj/std":0.3530284283574424,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/norm":0.0010347678359991972,"train/train/tensor_act_model_layers_12_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_k_proj/max_abs":5.21875,"train/train/tensor_act_model_layers_11_self_attn_q_proj/norm":6213.81459664947,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/std":3.6144057364393615e-05,"train/train/tensor_act_model_layers_77/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_v_proj/std":0.23291090761719274,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/mean":-1.1525116860866547e-07,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/mean":-0.000400543212890625,"train/train/tensor_act_model_layers_73_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn_o_proj/std":0.11914136339910524,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/norm":0.033083081623212045,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/norm":2.640625,"train/train/layer__model_layers_72/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/mean":1.2767850421369076e-07,"train/train/layer__model_layers_26/param/norm":19.81731753337848,"train/train/layer_model_layers_91/act/max_abs":15.75,"train/train/tensor_act_model_layers_4_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/std":0.22949220527042663,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/std":6.165266261435356e-05,"train/train/tensor_act_model_layers_17_mlp/max_abs":1.0390625,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_54/act/norm":13940.312949644538,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_post_attention_layernorm/norm":5792.611083985543,"train/train/tensor_act_model_layers_30_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/mean":5.51808625459671e-07,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/mean":-2.473592758178711e-06,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/std":4.1152843873532876e-05,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/std":5.668808355350547e-05,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/mean":3.066379576921463e-07,"train/train/tensor_act_model_layers_26_self_attn/norm":281.24286589023365,"train/train/tensor_act_model_layers_16_self_attn_v_proj/max_abs":1.6796875,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_input_layernorm/max_abs":5.78125,"train/train/tensor_act_model_layers_36/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/mean":-0.000293731689453125,"train/train/tensor_act_model_layers_21/norm":7865.07496498655,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/max_abs":0.000453948974609375,"train/train/tensor_act_model_layers_72_self_attn_o_proj/max_abs":1.34375,"train/train/tensor_act_model_layers_55_mlp_down_proj/std":0.09716858216671954,"train/train/tensor_act_model_layers_31_mlp_up_proj/norm":2874.996627953726,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_84_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/mean":3.781769919442013e-08,"train/train/layer__model_layers_38/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_82/grad/max_abs":0.00106048583984375,"train/train/tensor_act_model_layers_23_input_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_18/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/norm":7.65625,"train/train/tensor_param_model_layers_92_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_19_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_down_proj/mean":0.002960205078125,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/max_abs":0.00064849853515625,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/mean":-8.296221494674683e-06,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp/mean":-0.0095672607421875,"train/train/layer_model_layers_85/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/norm":9.6875,"train/train/tensor_act_model_layers_62_self_attn_q_proj/std":1.0625015707565366,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_19_mlp_up_proj/std":0.21875040871718235,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39/mean":0.079833984375,"train/train/layer_model_layers_5/grad/std":5.955531053525145e-05,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/max_abs":6.961822509765625e-05,"train/train/tensor_act_model_layers_60_mlp_down_proj/norm":629.2381183575887,"train/train/tensor_act_model_layers_64_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/mean":8.630752563476562e-05,"train/train/tensor_act_model_layers_10_input_layernorm/norm":5792.605468754764,"train/train/tensor_act_model_layers_9/norm":8425.371830738539,"train/train/tensor_act_model_layers_39_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/max_abs":0.23046875,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/max_abs":0.1611328125,"train/train/tensor_act_model_layers_32_self_attn_k_proj/mean":-0.05810546875,"train/train/tensor_act_model_layers_72_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_90/param/frac_near_user_limit":0,"train/train/layer_model_layers_44/grad/norm":0.04187728086182364,"train/train/tensor_act_model_layers_79_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/mean":2.2202730178833008e-06,"train/train/tensor_param_model_layers_86_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn/max_abs":0.75390625,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/mean":-6.670597940683365e-08,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/max_abs":0.000949859619140625,"train/train/tensor_act_model_layers_12_self_attn/max_abs":0.44140625,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/mean":3.036984708160162e-08,"train/train/tensor_act_model_layers_69_self_attn_v_proj/std":0.4047862133691136,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/norm":0.018006509001384226,"train/train/tensor_act_model_layers_7_mlp/mean":-0.002056121826171875,"train/train/tensor_act_model_layers_64_self_attn_v_proj/norm":2820.7002402212347,"train/train/tensor_act_model_layers_28/max_abs":17.875,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/max_abs":0.00030517578125,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/max_abs":0.0002651214599609375,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/mean":-1.1309981346130371e-05,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/max_abs":0.000274658203125,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/std":4.598840968431476e-05,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/norm":0.021729029960116948,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/max_abs":0.109375,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/norm":0.01525179893600367,"train/train/layer__model_layers_53/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/mean":-5.878973752260208e-08,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/std":0.037353515625,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn/mean":-0.0017528533935546875,"train/train/tensor_act_model_layers_28_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_k_proj/norm":5970.685796650592,"train/train/tensor_act_model_layers_76_self_attn_v_proj/norm":2452.395148563273,"train/train/layer__model_layers_3/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_14/act/max_abs":18.125,"train/train/layer__model_layers_23/param/max_abs":1,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/norm":0.0031520204330993955,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/std":0.0281982421875,"train/train/tensor_act_model_layers_93_mlp_up_proj/max_abs":6.5,"train/train/tensor_act_model_layers_37_self_attn_q_proj/max_abs":6.21875,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/max_abs":0.000606536865234375,"train/train/tensor_act_model_layers_39/max_abs":17.25,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/std":0.04638671875,"train/train/tensor_act_model_layers_12_input_layernorm/std":1.0000000229777581,"train/train/tensor_act_model_layers_67_self_attn_q_proj/max_abs":6.125,"train/train/tensor_act_model_layers_35_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_23/grad/mean":-4.232534194327368e-07,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/mean":-0.0002155303955078125,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/mean":3.859167918562889e-07,"train/train/tensor_act_model_layers_56/mean":0.05584716796875,"train/train/tensor_act_model_layers_46_self_attn_v_proj/std":0.36914070005769317,"train/train/tensor_act_model_layers_91_self_attn/mean":-0.0016422271728515625,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/max_abs":0.0001583099365234375,"train/train/tensor_param_model_layers_37_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/mean":0.00010347366333007812,"train/train/tensor_act_model_layers_21_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_post_attention_layernorm/mean":0.04974365234375,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/std":3.809848802155042e-05,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/max_abs":0.0001697540283203125,"train/train/tensor_act_model_layers_51_self_attn_k_proj/norm":4462.953685647249,"train/train/tensor_act_model_layers_2/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_o_proj/norm":600.7061882246599,"train/train/layer__model_layers_2/param/std":0.04740688019307652,"train/train/tensor_param_model_layers_74_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/max_abs":0.171875,"train/train/tensor_act_model_layers_16/frac_near_dtype_limit":0,"train/train/layer_model_layers_54/grad/std":5.167440201334705e-05,"train/train/tensor_param_model_layers_46_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn/max_abs":1.2890625,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/std":8.142802375290846e-05,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/max_abs":0.22265625,"train/train/tensor_act_model/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_o_proj/norm":454.1523125441613,"train/train/layer__model_layers_32/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/mean":9.655952453613281e-06,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp_up_proj/max_abs":2.296875,"train/train/tensor_act_model_layers_4_mlp_up_proj/norm":2889.980841444762,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_58/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_77/act/max_abs":12.5,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/max_abs":0.193359375,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/max_abs":0.0016326904296875,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/max_abs":0.0001010894775390625,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/std":0.037353515625,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_20_self_attn/std":0.022553727131681547,"train/train/layer__model_layers_66/param/frac_near_user_limit":0,"train/train/layer__model_layers_73/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/norm":0.006026391080772795,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_21/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_o_proj/norm":1005.9306311325272,"train/train/tensor_act_model_layers_28_mlp_down_proj/mean":0.00386810302734375,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/std":5.9788427345285296e-05,"train/train/tensor_act_model_layers_91_input_layernorm/norm":5792.607299808313,"train/train/tensor_act_model_layers_1_self_attn_k_proj/norm":7708.114220101754,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/norm":0.024495560993234276,"train/train/tensor_act_model_layers_41_self_attn_k_proj/mean":-0.00463104248046875,"train/train/layer_model_layers_56/act/max_abs":14.625,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/std":3.245654784492758e-05,"train/train/tensor_act_model_layers_0_self_attn_v_proj/max_abs":1.1953125,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/mean":4.2421743273735046e-07,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/mean":0.0002803802490234375,"train/train/tensor_act_model_layers_79_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/norm":3.453125,"train/train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_33/param/mean":0.0015113104524180968,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/max_abs":0.111328125,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/std":0.0291748046875,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/std":5.016945092842953e-05,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/norm":0.009752121171930177,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/std":0.049560546875,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/norm":0.02688770093873798,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/norm":0.0038644295428458923,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/max_abs":0.224609375,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/norm":4.96875,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/std":0.00010337921329865211,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_89/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/max_abs":0.1845703125,"train/train/tensor_act_model_layers_17_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_67_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_74_self_attn_k_proj/norm":4670.780439393191,"train/train/layer_model_layers_48/act/mean":-0.004414485051081731,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_41_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_1/grad/std":8.677278804192226e-05,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_v_proj/max_abs":2.21875,"train/train/tensor_act_model_layers_60_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_35/param/mean":0.0015780111184916146,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6/std":1.4883069061171428,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/mean":-0.0004100799560546875,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/max_abs":0.00014972686767578125,"train/train/tensor_act_model_layers_52_post_attention_layernorm/norm":5792.606323244085,"train/train/tensor_act_model_layers_88_mlp_up_proj/mean":-0.19091796875,"train/train/layer_model_layers_92/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/norm":0.0076839025742760775,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/max_abs":0.1201171875,"train/train/tensor_act_model_layers_26_mlp_up_proj/mean":-0.0687255859375,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/std":5.824356892287839e-05,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/norm":9.375,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_4/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/norm":0.010499732303612919,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/max_abs":0.228515625,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/max_abs":0.158203125,"train/train/tensor_param_model_layers_19_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_34_self_attn_v_proj/max_abs":2.3125,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/mean":1.0812655091285706e-06,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/norm":4.75,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_57_self_attn_q_proj/norm":5357.466643477769,"train/train/tensor_act_model_layers_83_self_attn_v_proj/norm":2809.2061585273914,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/norm":3.734375,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/norm":0.004471580259338639,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn/norm":427.94843800453395,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/std":0.02734375,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/std":5.323481492291142e-05,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/max_abs":0.25,"train/train/layer__model_layers_89/param/max_abs":1,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/std":2.5787488147113116e-05,"train/train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/mean":-2.175569534301758e-06,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/max_abs":0.1455078125,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/mean":0.00017547607421875,"train/train/tensor_act_model_layers_13_self_attn_o_proj/mean":-0.0007410049438476562,"train/train/layer__model_layers_4/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/std":0.023681640625,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/max_abs":0.00042724609375,"train/train/tensor_act_model_layers_36_mlp/norm":378.6767679017914,"train/train/tensor_act_model_layers_39_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/max_abs":0.000522613525390625,"train/train/tensor_act_model_layers_4/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/max_abs":0.0001392364501953125,"train/train/layer_model_layers_17/act/std":0.6925628204511236,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/norm":5.78125,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/std":0.034912109375,"train/train/tensor_act_model_layers_72/frac_near_dtype_limit":0,"train/train/layer__model_layers_68/param/max_abs":1,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/max_abs":0.000431060791015625,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_53_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_59_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7/std":1.4707277795066338,"train/train/tensor_act_model_layers_5_self_attn_o_proj/max_abs":0.75390625,"train/train/tensor_act_model_layers_72_self_attn_q_proj/max_abs":6.125,"train/train/layer_model_layers_9/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_mlp_down_proj/max_abs":1.4765625,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/std":3.531047217533365e-05,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_20/std":1.3633092587717142,"train/train/tensor_act_model_layers_50_mlp/max_abs":1.125,"train/train/tensor_act_model_layers_24_input_layernorm/norm":5792.608886720036,"train/train/tensor_act_model_layers_27_mlp_up_proj/mean":-0.065185546875,"train/train/tensor_act_model_layers_41/mean":0.0751953125,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/max_abs":0.1611328125,"train/train/layer_model_layers_68/act/norm":15157.00785584402,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_33_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/max_abs":0.1533203125,"train/train/tensor_param_model_layers_11_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_19/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/mean":-2.3831380531191826e-06,"train/train/tensor_act_model_layers_3_self_attn_o_proj/max_abs":1.2421875,"train/train/tensor_act_model_layers_78_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/std":0.041748046875,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/std":9.676792629201044e-06,"train/train/tensor_act_model_layers_9_post_attention_layernorm/max_abs":5.6875,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/norm":0.00957449802036515,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/mean":1.2200325727462769e-07,"train/train/tensor_act_model_layers_5/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/norm":2.828125,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/norm":4.25,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_9/param/max_abs":1,"train/train/tensor_act_model_layers_84_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/std":3.412363859600538e-05,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/std":4.6040054035253336e-05,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/mean":-0.00019168853759765625,"train/train/tensor_act_model_layers_38_self_attn_v_proj/std":0.39453130862721275,"train/train/tensor_act_model_layers_60_self_attn_k_proj/norm":5324.509342845301,"train/train/tensor_act_model_layers_10_post_attention_layernorm/max_abs":5.6875,"train/train/tensor_act_model_layers_77/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_input_layernorm/norm":5792.608642579715,"train/train/tensor_act_model_layers_59_mlp/std":0.1077883408263704,"train/train/tensor_act_model_layers_38_mlp_up_proj/mean":-0.07666015625,"train/train/tensor_param_model_layers_54_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/mean":2.6420457288622856e-07,"train/train/tensor_act_model_layers_2_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp/max_abs":2.15625,"train/train/tensor_act_model_layers_74_mlp_up_proj/std":0.4980472789089584,"train/train/tensor_act_model_layers_11_self_attn_v_proj/max_abs":1.453125,"train/train/tensor_act_model_layers_29_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_v_proj/max_abs":3.078125,"train/train/tensor_act_model_layers_73_post_attention_layernorm/mean":0.037841796875,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18/max_abs":17.875,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/std":3.6902235009066013e-05,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24/max_abs":17.75,"train/train/tensor_act_model_layers_74_input_layernorm/std":1.0000014081587414,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_input_layernorm/std":0.9960940491918975,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/std":2.692658241309136e-05,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/norm":0.0017361734568640133,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_27_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp_up_proj/std":0.2285163422923772,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/mean":-1.0454095900058746e-07,"train/train/tensor_act_model_layers_64/max_abs":13.4375,"train/train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/std":0.0593261722429299,"train/train/tensor_act_model_layers_23_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn_o_proj/max_abs":0.62109375,"train/train/tensor_act_model_layers_16/frac_near_user_limit":0,"train/train/layer_model_layers_69/act/std":0.7215667952164136,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_v_proj/max_abs":2.9375,"train/train/tensor_act_model_layers_17_post_attention_layernorm/norm":5792.601562508859,"train/train/tensor_param_model_layers_82_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_input_layernorm/max_abs":6.09375,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/norm":0.0009687772760714028,"train/train/tensor_act_model_layers_67_self_attn_o_proj/max_abs":1.3359375,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/mean":4.298985004425049e-06,"train/train/tensor_act_model_layers_24_self_attn_q_proj/std":1.0937500068119594,"train/train/layer_model_layers_43/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp_up_proj/max_abs":3.125,"train/train/tensor_act_model_layers_52_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/std":6.652792011030606e-05,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_81/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/std":0.00013543038870363476,"train/train/tensor_act_model_layers_61/norm":7848.54419349401,"train/train/global/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/std":0.00010914414362078552,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/max_abs":0.11328125,"train/train/tensor_act_model_layers_25_self_attn_k_proj/std":0.8642595883798404,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_v_proj/max_abs":2.21875,"train/train/tensor_act_model_layers_38_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/max_abs":0.0003910064697265625,"train/train/tensor_act_model_layers_64_post_attention_layernorm/std":1.000000081956383,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/mean":0.000461578369140625,"train/train/tensor_act_model_layers_13/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_v_proj/std":0.38720819062625234,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/std":0.0001146605844795854,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/max_abs":0.00063323974609375,"train/train/layer_model_layers_17/act/norm":14464.255866876672,"train/train/tensor_act_model_layers_67/norm":8390.213349913896,"train/train/tensor_act_model_layers_70_mlp_down_proj/std":0.15820483139221242,"train/train/layer_model_layers_19/grad/max_abs":0.00084686279296875,"train/train/tensor_act_model_layers_78_self_attn_q_proj/std":0.9755880085652986,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/std":0.0250244140625,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/std":0.025390625,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/norm":10.1875,"train/train/tensor_param_model_layers_73_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp/max_abs":0.85546875,"train/train/tensor_act_model_layers_30/norm":7638.484817035184,"train/train/tensor_act_model_layers_2_input_layernorm/mean":0.028472900390625,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp_down_proj/max_abs":0.388671875,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/max_abs":0.0003261566162109375,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/max_abs":8.440017700195312e-05,"train/train/tensor_act_model_layers_36_input_layernorm/mean":0.0789794921875,"train/train/tensor_act_model_layers_78_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/max_abs":0.2490234375,"train/train/tensor_act_model_layers_19_mlp_down_proj/mean":0.006103515625,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_input_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_65/param/max_abs":1,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/mean":-4.359753802418709e-08,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/max_abs":0.00177001953125,"train/train/tensor_act_model_layers_24/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/norm":0.0024496772852508394,"train/train/layer_model_layers_80/grad/max_abs":0.00162506103515625,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_input_layernorm/std":1.0000000204890966,"train/train/tensor_act_model_layers_3_self_attn_o_proj/mean":-0.00211334228515625,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/std":1.675584317925996e-05,"train/train/tensor_act_model_layers_10_self_attn_k_proj/norm":6444.800180492653,"train/train/tensor_act_model_layers_93_self_attn/mean":-0.00699615478515625,"train/train/tensor_act_model_layers_81_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp_up_proj/mean":-0.051025390625,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_63/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/std":0.057373046875,"train/train/tensor_act_model_layers_44_self_attn_k_proj/norm":5073.79680275236,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/max_abs":0.228515625,"train/train/tensor_act_model_layers_74_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_mlp/mean":-0.00921630859375,"train/train/tensor_act_model_layers_68_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/mean":8.800998330116272e-08,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/mean":1.5081604942679405e-07,"train/train/layer_model_layers_60/grad/std":6.815811557888259e-05,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/std":9.508976967559349e-05,"train/train/tensor_act_model_layers_86_mlp_down_proj/mean":0.0114288330078125,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/std":0.0419921875,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/norm":0.03381457826217423,"train/train/tensor_param_model_layers_13_input_layernorm_weight/mean":1,"train/train/layer__model_layers_71/param/std":0.055659624537765115,"train/train/tensor_act_model_layers_16_self_attn/std":0.04046695038088668,"train/train/layer__model_layers_55/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/norm":0.006555155180431651,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/std":4.959285642712254e-05,"train/train/layer_model_layers_50/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/max_abs":0.130859375,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/norm":0.008668979900394672,"train/train/layer_model_layers_74/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_up_proj/mean":-0.05206298828125,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/max_abs":0.00055694580078125,"train/train/tensor_act_model_layers_46_self_attn_q_proj/mean":0.011138916015625,"train/train/tensor_act_model_layers_20_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_84/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/mean":5.245208740234375e-06,"train/train/tensor_act_model_layers_53_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/std":9.563749237438098e-05,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_39/grad/mean":-7.864599611592739e-07,"train/train/tensor_act_model_layers_20_post_attention_layernorm/max_abs":6,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/mean":2.670858521014452e-07,"train/train/tensor_act_model_layers_53_self_attn/norm":484.48595931229715,"train/train/tensor_act_model_layers_72_self_attn/norm":447.84223212527223,"train/train/tensor_act_model_layers_77_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn/mean":-0.000195235013961792,"train/train/tensor_act_model_layers_40_input_layernorm/max_abs":6.03125,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/norm":6.34375,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/max_abs":6.771087646484375e-05,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/std":1.9954133493102395e-05,"train/train/layer__model_layers_55/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_o_proj/mean":-0.0015048980712890625,"train/train/tensor_act_model_layers_81_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/norm":5.3125,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/mean":5.710870027542114e-06,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/norm":0.01173217322548214,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/std":0.041748046875,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/std":2.4916186824374967e-05,"train/train/tensor_act_model_layers_50_mlp_up_proj/norm":3884.6065124722754,"train/train/tensor_param_model_layers_93_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_lm_head/norm":100331.65184566984,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_o_proj/norm":251.28645741320037,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/mean":-3.4371623769402504e-07,"train/train/tensor_act_model_layers_67_mlp_down_proj/max_abs":1.28125,"train/train/tensor_act_model_layers_31_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/mean":5.117617547512054e-07,"train/train/tensor_act_model_layers_3_self_attn_k_proj/mean":-0.06207275390625,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/norm":6.96875,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/std":0.04248046875,"train/train/layer_model_layers_79/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn_k_proj/mean":-0.1138916015625,"train/train/layer_model_layers_73/grad/frac_near_user_limit":0,"train/train/layer_model_layers_35/grad/norm":0.04050165368069781,"train/train/tensor_act_model_layers_72_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_post_attention_layernorm/mean":0.0318603515625,"train/train/tensor_act_model_layers_2_mlp_up_proj/mean":-0.05145263671875,"train/train/tensor_param_model_layers_89_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/std":0.0191650390625,"train/train/layer_model_layers_92/grad/norm":0.10405090333832567,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/std":0.05615234375,"train/train/tensor_act_model_layers_14_input_layernorm/max_abs":6,"train/train/tensor_act_model_layers_77_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_post_attention_layernorm/norm":5792.608520509586,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_post_attention_layernorm/std":1.000000094994898,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/max_abs":0.00010156631469726562,"train/train/tensor_act_model_layers_43_self_attn_v_proj/max_abs":2.796875,"train/train/layer__model_layers_19/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/norm":8.9375,"train/train/tensor_act_model_layers_90_post_attention_layernorm/norm":5792.608886722609,"train/train/tensor_act_model_layers_87_input_layernorm/std":0.9960957471042134,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/norm":0.0004580377711785044,"train/train/tensor_act_model_layers_31_self_attn_v_proj/norm":1837.526111348826,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/std":6.929515719333136e-06,"train/train/tensor_act_model_layers_64_mlp_up_proj/std":0.4628913171175519,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/norm":5.40625,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/norm":0.0030355116422545315,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/norm":0.027924767974599928,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_k_proj/std":0.9189469948569778,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/max_abs":0.000133514404296875,"train/train/tensor_param_model_layers_67_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/norm":5.6875,"train/train/layer_model_layers_7/act/std":0.846261559220837,"train/train/tensor_param_model_layers_15_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_10_mlp/norm":255.64503033449438,"train/train/tensor_act_model_layers_28_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp_down_proj/mean":0.00390625,"train/train/tensor_act_model_layers_23_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/mean":7.511698640882969e-08,"train/train/tensor_act_model_layers_88_mlp/max_abs":3.9375,"train/train/tensor_act_model_layers_69_post_attention_layernorm/std":1.000001421197239,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/norm":0.019414361226107325,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/mean":2.6618363335728645e-07,"train/train/tensor_act_model_layers_40_self_attn_o_proj/mean":-0.0017528533935546875,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_27_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer__model_layers_73/param/max_abs":1,"train/train/layer__model_layers_31/param/std":0.050015570661029754,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/max_abs":0.25390625,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/norm":0.003017664247660334,"train/train/tensor_param_model_layers_48_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/max_abs":0.00021648406982421875,"train/train/tensor_act_model_layers_1/std":1.4961192923175697,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/max_abs":0.000438690185546875,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn/norm":525.9160700821067,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/norm":0.02492193311726362,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/mean":5.634574335999787e-07,"train/train/tensor_act_model_layers_85_self_attn_v_proj/norm":2543.5596694757337,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_v_proj/std":0.5107451261570597,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/std":2.2884740661394544e-05,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/max_abs":0.00018215179443359375,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/mean":9.284121915698051e-08,"train/train/tensor_act_model_layers_55/norm":7658.022898679712,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/mean":0.000286102294921875,"train/train/tensor_param_model_layers_34_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_45_self_attn_q_proj/norm":5596.803536548339,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/std":0.0380859375,"train/train/tensor_act_model_layers_2_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/mean":-0.0007781982421875,"train/train/tensor_act_model_layers_59_post_attention_layernorm/max_abs":5.875,"train/train/tensor_act_model_layers_80_self_attn_q_proj/std":1.2148503429074489,"train/train/layer__model_layers_81/param/mean":0.0013938136108208,"train/train/tensor_act_model_layers_12_self_attn_k_proj/mean":-0.01226806640625,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/mean":0.001312255859375,"train/train/tensor_param_model_layers_40_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/norm":0.0009928917609996271,"train/train/tensor_act_model_layers_2/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/norm":6,"train/train/layer_model_layers_68/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_35/grad/max_abs":0.00147247314453125,"train/train/tensor_act_model_layers_34_input_layernorm/std":0.9960938322777808,"train/train/tensor_act_model_layers_67_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/mean":-2.1774321794509888e-06,"train/train/tensor_act_model_layers_91_mlp_up_proj/mean":-0.15087890625,"train/train/layer_model_layers_28/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/std":0.00012038412354553592,"train/train/layer_model_layers_74/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_12/grad/std":3.0059178152740056e-05,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/max_abs":0.000820159912109375,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/mean":3.3527612686157227e-07,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/mean":-8.218921720981598e-08,"train/train/tensor_act_model_embed_tokens/max_abs":0.5,"train/train/tensor_act_model_layers_56_input_layernorm/mean":0.048095703125,"train/train/tensor_act_model_layers_16_mlp_down_proj/std":0.03216570559168148,"train/train/tensor_act_model_layers_51_mlp_up_proj/norm":3903.74793063851,"train/train/tensor_act_model_layers_41_mlp_up_proj/max_abs":4.25,"train/train/tensor_act_model_layers_2_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/mean":-7.12275505065918e-06,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/mean":-0.0002193450927734375,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/std":0.0419921875,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/norm":6.21875,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp_down_proj/mean":-0.0025482177734375,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/norm":0.012954846708317614,"train/train/tensor_act_model_layers_59_input_layernorm/max_abs":5.84375,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/max_abs":0.00019931793212890625,"train/train/tensor_act_model_layers_34_mlp_up_proj/std":0.2851564590244311,"train/train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/std":0.027587890625,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/std":3.688055382188612e-05,"train/train/tensor_act_model_layers_73/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/std":2.0454535515616173e-05,"train/train/tensor_act_model_layers_48_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_input_layernorm/mean":0.0445556640625,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/mean":6.063783075660467e-08,"train/train/tensor_act_model_layers_38_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/max_abs":0.125,"train/train/tensor_param_model_layers_39_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/std":3.066295490846726e-05,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/norm":3.59375,"train/train/tensor_act_model_layers_1_self_attn/max_abs":0.6640625,"train/train/tensor_act_model_layers_49_input_layernorm/mean":0.05859375,"train/train/tensor_act_model_layers_44_mlp_up_proj/std":0.34277508944469276,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_mlp_down_proj/max_abs":0.71875,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_90/act/mean":-0.0042667388916015625,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/norm":0.02197435160560223,"train/train/tensor_act_model_layers_3_self_attn/std":0.12451174703298509,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_21/grad/mean":-3.9678055708777516e-07,"train/train/tensor_act_model_layers_68_self_attn_k_proj/std":0.8378931429798466,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/max_abs":0.00194549560546875,"train/train/layer_model_layers_68/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_o_proj/mean":-0.0002765655517578125,"train/train/tensor_act_model_layers_45_mlp_up_proj/max_abs":4.125,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/mean":1.1245720088481903e-07,"train/train/layer_model_layers_41/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/max_abs":0.00070953369140625,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/std":0.0341796875,"train/train/tensor_param_model_layers_45_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn_v_proj/mean":-0.0008559226989746094,"train/train/tensor_act_model_layers_16_self_attn_q_proj/norm":5274.348904787382,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/mean":-0.00099945068359375,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_38/act/std":0.6833689087698839,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/max_abs":0.0002117156982421875,"train/train/layer__model_layers_13/param/norm":19.696903756126876,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/max_abs":0.000583648681640625,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/std":0.025146484375,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/std":4.2444889967185615e-05,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_post_attention_layernorm/mean":0.0430908203125,"train/train/tensor_act_model_layers_21_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/mean":-3.043562173843384e-06,"train/train/tensor_act_model_layers_54_self_attn/mean":-0.00235748291015625,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/max_abs":0.1357421875,"train/train/tensor_act_model_layers_9_self_attn_o_proj/std":0.18019368434800248,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp/mean":-0.0011882781982421875,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/frac_near_dtype_limit":0,"train_samples_per_second":206.491,"train/train/tensor_act_model_layers_22_mlp_down_proj/std":0.044250955743622505,"train/train/layer__model_layers_8/param/std":0.048709806980437635,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/std":3.3428924407390186e-05,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/max_abs":0.00018787384033203125,"train/train/layer_model_layers_48/act/norm":14152.366141103485,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/norm":6.65625,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_93/act/frac_near_dtype_limit":0,"train/train/layer__model_layers_73/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp_down_proj/max_abs":0.55859375,"train/train/tensor_param_model_layers_2_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/norm":0.007295105611384409,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/std":0.055908203125,"train/train/layer__model_layers_78/param/max_abs":1,"train/train/tensor_act_model_layers_8_mlp_up_proj/max_abs":2.625,"train/train/tensor_act_model_layers_22_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/std":7.796915114317302e-05,"train/train/tensor_act_model_layers_52_mlp/max_abs":1.0234375,"train/train/layer_model_layers_59/grad/max_abs":0.001220703125,"train/train/tensor_act_model_layers_6_mlp/mean":-0.00548553466796875,"train/train/tensor_act_model_layers_75_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_31/act/max_abs":17.75,"train/train/tensor_act_model_layers_69_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/std":0.0234375,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/std":0.028076171875,"train/train/tensor_act_model_layers_58_mlp_up_proj/std":0.4082031980085536,"train/train/tensor_act_model_layers_35_mlp/norm":285.1341781058447,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/mean":-8.888542652130127e-06,"train/train/layer_model_layers_73/grad/std":7.216296416767486e-05,"train/train/layer_model_layers_60/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/max_abs":0.0008544921875,"train/train/tensor_act_model_layers_24_mlp/max_abs":1.1015625,"train/train/tensor_param_model_layers_72_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/std":0.03564453125,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/max_abs":0.2001953125,"train/train/tensor_act_model_layers_32_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/max_abs":0.1259765625,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/mean":3.495370037853718e-08,"train/train/tensor_act_model_layers_14_self_attn_v_proj/max_abs":2.171875,"train/train/tensor_act_model_layers_42_self_attn_o_proj/std":0.06774943701729534,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/norm":0.03183725038407374,"train/train/tensor_act_model_layers_22_self_attn/std":0.06281032948901312,"train/train/tensor_act_model_layers_14_self_attn/norm":297.7368160571121,"train/train/tensor_act_model_layers_75_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_mlp_down_proj/mean":-0.0001195669174194336,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/mean":6.407499313354492e-06,"train/train/tensor_act_model_layers_49_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/norm":8.5,"train/train/tensor_act_model_layers_46_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/max_abs":0.00010347366333007812,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/norm":5.78125,"train/train/tensor_act_model_layers_23_input_layernorm/norm":5792.608642580445,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/norm":0.0038429510931030867,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn_v_proj/mean":-0.008636474609375,"train/train/layer_model_layers_8/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_v_proj/max_abs":3.40625,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/std":6.401671529526069e-05,"train/train/tensor_act_model_layers_11_post_attention_layernorm/max_abs":5.875,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/std":1.3245208852151027e-05,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/mean":-2.028420567512512e-06,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/mean":-0.0003452301025390625,"train/train/layer_model_layers_86/grad/frac_near_user_limit":0,"train/train/layer_model_layers_6/act/max_abs":18.75,"train/train/layer_model_layers_17/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79/max_abs":11.8125,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/norm":0.0025522120801732924,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/std":6.881427193562304e-05,"train/train/tensor_act_model_layers_43_self_attn_v_proj/std":0.42578126661000965,"train/train/tensor_act_model_layers_89_mlp_up_proj/std":0.7539063685916155,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/mean":-4.123896360397339e-06,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/norm":0.013628037881780491,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/mean":1.4822580851614475e-07,"train/train/global/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/std":1.5628260928373184e-05,"train/train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer__model_layers_71/param/mean":0.0012457218259433502,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/std":5.146096872599553e-05,"train/train/tensor_act_model_layers_41/norm":7541.044355161032,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/mean":-2.5960616767406464e-07,"train/train/tensor_param_model_layers_38_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/mean":1.1129304766654968e-06,"train/train/tensor_act_model_layers_76_input_layernorm/max_abs":5.5625,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/mean":5.1800161600112915e-06,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp_up_proj/norm":4275.20967243544,"train/train/layer_model_layers_83/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/max_abs":0.1015625,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_85/act/norm":17260.887568748163,"train/train/tensor_act_model_layers_43_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_post_attention_layernorm/std":1.0000000447034827,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/norm":10.8125,"train/train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/std":0.0284423828125,"train/train/tensor_act_model_layers_53_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_v_proj/max_abs":2.65625,"train/train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp/norm":1086.6291796583819,"train/train/tensor_act_model_layers_12_mlp/std":0.040283203666860404,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/mean":0.000453948974609375,"train/train/tensor_act_model_layers_70_self_attn_v_proj/norm":3441.1745749065635,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/std":4.8481254485750946e-05,"train/train/layer__model_layers_80/param/std":0.058693134890120985,"train/train/tensor_act_model_layers_51_self_attn_v_proj/mean":-0.0069427490234375,"train/train/tensor_act_model_layers_51_mlp_up_proj/max_abs":2.96875,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/std":0.04345703125,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/max_abs":0.00023937225341796875,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/norm":0.0004300679952418568,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/mean":-1.2973323464393616e-06,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn_k_proj/mean":0.04779052734375,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_51/grad/mean":-8.701037342388433e-07,"train/train/tensor_act_model_layers_19_post_attention_layernorm/std":1.000000034924596,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/mean":-1.7657876014709473e-06,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_93/param/std":0.06438776109286039,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_v_proj/max_abs":1.703125,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/std":4.737869723069608e-05,"train/train/tensor_act_model_layers_14_mlp/max_abs":0.39453125,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/mean":5.662441253662109e-06,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/mean":-3.9577484130859375e-05,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/mean":6.742775440216064e-07,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/norm":3.75,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/mean":0.00012493133544921875,"train/train/tensor_param_model_layers_66_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_71_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp_up_proj/max_abs":2.3125,"train/train/tensor_act_model_layers_69/std":1.460949647822338,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn/mean":-0.00034880638122558594,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp_down_proj/max_abs":1.03125,"train/train/tensor_act_model_layers_62_mlp/max_abs":1.0234375,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_q_proj/mean":-0.03997802734375,"train/train/tensor_act_model_layers_4_post_attention_layernorm/max_abs":5,"train/train/tensor_param_model_layers_15_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/max_abs":0.0018768310546875,"train/train/tensor_act_model_layers_34_mlp/max_abs":0.69140625,"train/train/tensor_param_model_layers_80_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/std":0.029296875,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/mean":3.043562173843384e-06,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp/std":0.31347809996183323,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/mean":0.00012302398681640625,"train/train/tensor_act_model_layers_66_self_attn/std":0.20581317208535532,"train/train/tensor_act_model_layers_59/max_abs":14.1875,"train/train/tensor_act_model_layers_3_mlp_up_proj/max_abs":4.5625,"train/train/tensor_act_model_layers_14_mlp/std":0.04614258064794786,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/mean":2.0116567611694336e-07,"train/train/layer_model_layers_20/act/norm":13325.747470133701,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/max_abs":0.0002613067626953125,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_input_layernorm/norm":5792.606323245957,"train/train/tensor_param_model_layers_73_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_87_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/max_abs":0.1689453125,"eval/steps_per_second":4.604,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/mean":4.3015461415052414e-08,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/norm":0.023481090064642562,"train/train/tensor_act_model_layers_41_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/std":5.986950327154737e-05,"train/train/layer__model_layers_74/param/mean":0.0016293815815132606,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/mean":-0.000476837158203125,"train/train/tensor_act_model_layers_76/std":1.591804752447377,"train/train/tensor_act_model_layers_45_mlp/norm":462.3028784268603,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_input_layernorm/std":1.0000000819563832,"train/train/tensor_act_model_layers_81_self_attn_v_proj/mean":9.608268737792969e-05,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/max_abs":0.216796875,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/norm":0.026139441363619335,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_49/grad/norm":0.041373440765125835,"train/train/tensor_act_model_layers_69_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52/max_abs":15.4375,"train/train/tensor_act_model_layers_78_mlp/mean":0.018646240234375,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/norm":0.031096917954281912,"train/train/layer__model_layers_7/param/norm":19.94221388239656,"train/train/tensor_act_model_layers_50_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/max_abs":0.001007080078125,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_down_proj/max_abs":0.9296875,"train/train/tensor_act_model_layers_66_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/norm":0.0009282774026148419,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_84_mlp_up_proj/max_abs":3.953125,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/mean":-0.000270843505859375,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/norm":0.0013478465049625548,"train/train/layer_model_layers_1/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/mean":0.0003833770751953125,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/std":0.034423828125,"train/train/tensor_act_model_layers_1_mlp/std":0.18774461730653447,"train/train/layer_model_layers_60/act/norm":14705.665761279482,"train/train/layer_model_layers_42/grad/norm":0.04066728884099797,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_39/act/std":0.659202298079628,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/max_abs":0.142578125,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/max_abs":0.1083984375,"train/train/tensor_act_model_layers_91_self_attn_o_proj/std":0.24194785508573846,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_input_layernorm/norm":5792.606323245134,"train/train/tensor_act_model_layers_88_post_attention_layernorm/std":0.9960953880745378,"train/train/tensor_act_model_layers_66_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80/mean":0.0950927734375,"train/train/tensor_act_model_layers_52_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/norm":0.0006322163044224281,"train/train/tensor_act_model_layers_80_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_48/param/norm":21.457750548315868,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_17/param/std":0.04846275009775069,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/norm":4.78125,"train/train/tensor_act_model_layers_2_mlp_up_proj/std":0.2915051522539816,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/std":0.041259765625,"train/train/tensor_act_model_layers_35_self_attn_o_proj/norm":719.8087211490051,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/max_abs":0.150390625,"train/train/tensor_act_model_layers_44_self_attn/frac_near_dtype_limit":0,"train/train/layer__model_layers_21/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35/norm":7610.224736207985,"train/train/tensor_act_model_layers_20_input_layernorm/max_abs":6,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/norm":2.921875,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/std":2.151124285659531e-05,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/mean":2.5262124836444855e-08,"train/train/tensor_act_model_layers_61_input_layernorm/max_abs":5.90625,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/std":0.0225830078125,"train/train/tensor_param_model_layers_23_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/std":0.03759765625,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/norm":0.049262744677811277,"train/train/tensor_act_model_layers_82_mlp_down_proj/mean":0.01336669921875,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/mean":-0.000652313232421875,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp_up_proj/mean":-0.075927734375,"train/train/tensor_act_model_layers_16_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/max_abs":0.000362396240234375,"train/train/tensor_act_model_layers_29_mlp/norm":424.34578283162824,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_post_attention_layernorm/norm":5792.612426758259,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/std":3.10457173554526e-05,"train/train/tensor_act_model_layers_47_self_attn_q_proj/norm":5877.616489027821,"train/train/tensor_act_model_layers_89_mlp/norm":2591.8679217349736,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_24/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/std":0.0234375,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_74/act/max_abs":12.9375,"train/train/layer_model_layers_4/act/mean":-0.01349654564490685,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn/mean":0.00189208984375,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/norm":0.025991017041512403,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/max_abs":0.31640625,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/norm":6.15625,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/max_abs":0.00116729736328125,"train/train/layer__model_layers_93/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/norm":0.008225110088824747,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/norm":4.71875,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/std":8.431453441651433e-05,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/std":2.0821955608364127e-05,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/max_abs":0.2001953125,"train/train/layer__model_layers_12/param/std":0.047288211928560045,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp_down_proj/mean":0.013885498046875,"train/train/tensor_act_model_layers_13_self_attn_k_proj/max_abs":4.625,"train/train/layer__model_layers_33/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/std":0.0277099609375,"train/train/layer__model_layers_0/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_down_proj/norm":916.2282459230611,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/norm":0.006249307802731581,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn/max_abs":1.9609375,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/std":4.563108554597022e-05,"train/train/tensor_act_model_layers_70_mlp_up_proj/norm":5120.818831319105,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/norm":0.017540998458216776,"train/train/layer__model_layers_41/param/norm":20.66772326978107,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/std":0.00010586598964088145,"train/train/layer__model_layers_35/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_q_proj/norm":4759.11574915503,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/max_abs":0.16015625,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/max_abs":0.1591796875,"train/train/time_per_step_avg":1.9891077641025185,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/norm":0.0035020084403329208,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_84/grad/mean":1.3321298123707824e-06,"train/train/layer_model_layers_39/grad/max_abs":0.0017242431640625,"train/train/layer_model_layers_50/act/std":0.6625514713571143,"train/train/tensor_act_model_layers_4_self_attn_q_proj/norm":6459.024721276636,"train/train/tensor_act_model_layers_26_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/mean":6.064772605895996e-06,"train/train/tensor_act_model_layers_10_self_attn_v_proj/mean":-0.003963470458984375,"train/train/tensor_act_model_layers_57/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_32_self_attn_o_proj/mean":-2.8818845748901367e-05,"train/train/tensor_act_model_layers_44_self_attn_v_proj/mean":0.002227783203125,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/frac_near_user_limit":0} \ No newline at end of file diff --git a/wandb/run-20260809_035819-cvzjg5ej/logs/debug-core.log b/wandb/run-20260809_035819-cvzjg5ej/logs/debug-core.log new file mode 100644 index 0000000000000000000000000000000000000000..6ab6c95a5b7718c427082e9937cc663568192adb --- /dev/null +++ b/wandb/run-20260809_035819-cvzjg5ej/logs/debug-core.log @@ -0,0 +1,58 @@ +{"time":"2026-08-09T03:56:53.721154509Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmpyqomfhgp/port-2869678.txt","pid":2869678,"detached":false,"idle-timeout":600000000000,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false} +{"time":"2026-08-09T03:56:53.722258921Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":2869678} +{"time":"2026-08-09T03:56:53.722234577Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-2869678-2909401-2521299324/socket","Net":"unix"}} +{"time":"2026-08-09T03:56:53.900147851Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"} +{"time":"2026-08-09T03:58:19.204299995Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"2(@)"} +{"time":"2026-08-09T03:58:19.282614427Z","level":"INFO","msg":"handleInformInit: received","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T03:58:19.544211049Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T03:58:24.897517362Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"xv8tfjtqu29x"} +{"time":"2026-08-09T05:00:38.67636271Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"xv8tfjtqu29x"} +{"time":"2026-08-09T05:00:40.520576799Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"xv8tfjtqu29x"} +{"time":"2026-08-09T05:00:40.552184373Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T05:00:40.553143275Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608126464Z","level":"INFO","msg":"processOutgoingData: finished","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608106419Z","level":"INFO","msg":"connection: closing","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608216462Z","level":"INFO","msg":"connection: closed successfully","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608222262Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"2(@)"} +{"time":"2026-08-09T05:00:50.241711721Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"3(@)"} +{"time":"2026-08-09T05:00:50.316234404Z","level":"INFO","msg":"handleInformInit: received","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:00:50.57530289Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:00:55.905488223Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"4jxnuihlia82"} +{"time":"2026-08-09T05:57:13.765576248Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"4jxnuihlia82"} +{"time":"2026-08-09T05:57:15.666017984Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"4jxnuihlia82"} +{"time":"2026-08-09T05:57:15.999725293Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:57:16.001240777Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068849513Z","level":"INFO","msg":"connection: closing","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068936758Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068853914Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068948948Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"} +{"time":"2026-08-09T05:57:26.287713024Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"4(@)"} +{"time":"2026-08-09T05:57:26.365087395Z","level":"INFO","msg":"handleInformInit: received","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T05:57:26.623707494Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T05:57:31.962348796Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"rz2ldpq16nht"} +{"time":"2026-08-09T07:02:02.153459338Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"rz2ldpq16nht"} +{"time":"2026-08-09T07:02:04.144735717Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"rz2ldpq16nht"} +{"time":"2026-08-09T07:02:04.180963791Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T07:02:04.181861619Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233097548Z","level":"INFO","msg":"connection: closing","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233183816Z","level":"INFO","msg":"connection: closed successfully","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233105065Z","level":"INFO","msg":"processOutgoingData: finished","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233192773Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"4(@)"} +{"time":"2026-08-09T07:02:13.885910885Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"5(@)"} +{"time":"2026-08-09T07:02:13.954769181Z","level":"INFO","msg":"handleInformInit: received","streamId":"uvqyddz0","id":"5(@)"} +{"time":"2026-08-09T07:02:14.215272022Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"uvqyddz0","id":"5(@)"} +{"time":"2026-08-09T07:02:19.530617395Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"fk71ydt1h8f7"} +{"time":"2026-08-09T07:03:39.9336748Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"fk71ydt1h8f7"} +{"time":"2026-08-09T07:03:40.29901782Z","level":"INFO","msg":"connection: closing","id":"5(@)"} +{"time":"2026-08-09T07:03:40.299111085Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"} +{"time":"2026-08-09T07:03:40.299007456Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"} +{"time":"2026-08-09T07:03:40.299122073Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"} +{"time":"2026-08-09T07:03:42.349833633Z","level":"INFO","msg":"connection: closing","id":"1(@)"} +{"time":"2026-08-09T07:03:42.349942779Z","level":"INFO","msg":"connection: closed successfully","id":"1(@)"} +{"time":"2026-08-09T07:03:42.34985531Z","level":"INFO","msg":"processOutgoingData: finished","id":"1(@)"} +{"time":"2026-08-09T07:03:42.349954994Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"1(@)"} +{"time":"2026-08-09T07:03:42.353940558Z","level":"INFO","msg":"server: parent process exited, terminating service process"} +{"time":"2026-08-09T07:03:42.353988499Z","level":"INFO","msg":"server: is shutting down"} +{"time":"2026-08-09T07:03:42.354128892Z","level":"INFO","msg":"server: listener closed","addr":{"Name":"/tmp/wandb-2869678-2909401-2521299324/socket","Net":"unix"}} +{"time":"2026-08-09T07:03:42.354189401Z","level":"INFO","msg":"server: forced shutdown"} +{"time":"2026-08-09T07:03:42.354198506Z","level":"ERROR","msg":"main: Serve() returned error","error":"forced shutdown"} diff --git a/wandb/run-20260809_035819-cvzjg5ej/logs/debug-internal.log b/wandb/run-20260809_035819-cvzjg5ej/logs/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..917956c7b486c0d5d39d6c44690689245c99e868 --- /dev/null +++ b/wandb/run-20260809_035819-cvzjg5ej/logs/debug-internal.log @@ -0,0 +1,535 @@ +{"time":"2026-08-09T03:58:19.282783063Z","level":"INFO","msg":"wandb-core"} +{"time":"2026-08-09T03:58:19.283092454Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"} +{"time":"2026-08-09T03:58:19.544029023Z","level":"INFO","msg":"stream: created new stream","id":"cvzjg5ej"} +{"time":"2026-08-09T03:58:19.544113293Z","level":"INFO","msg":"handler: started"} +{"time":"2026-08-09T03:58:19.544204994Z","level":"INFO","msg":"stream: started"} +{"time":"2026-08-09T03:58:19.544224319Z","level":"INFO","msg":"writer: started","stream_id":"cvzjg5ej"} +{"time":"2026-08-09T03:58:19.544236358Z","level":"INFO","msg":"sender: started"} +{"time":"2026-08-09T03:58:20.49399574Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1} +{"time":"2026-08-09T03:58:20.585402614Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T03:58:35.494346021Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":0,"events_lines":2,"console_offset":1,"console_lines":4,"uploaded_len":2} +{"time":"2026-08-09T03:58:35.60993089Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T03:58:50.494317925Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":2,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T03:58:50.598887827Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T03:59:03.877770359Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":430} +{"time":"2026-08-09T03:59:03.878186402Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":15} +{"time":"2026-08-09T03:59:03.883932073Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":2508} +{"time":"2026-08-09T03:59:03.883962742Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1} +{"time":"2026-08-09T03:59:03.894578099Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":5231} +{"time":"2026-08-09T03:59:03.89461253Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1} +{"time":"2026-08-09T03:59:03.896155708Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":5757} +{"time":"2026-08-09T03:59:03.911398768Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1574} +{"time":"2026-08-09T03:59:03.912624837Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":7504} +{"time":"2026-08-09T03:59:03.912762308Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":23} +{"time":"2026-08-09T03:59:03.930332533Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":11798} +{"time":"2026-08-09T03:59:03.933304856Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":220} +{"time":"2026-08-09T03:59:03.934700228Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":12563} +{"time":"2026-08-09T03:59:03.93682743Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":829} +{"time":"2026-08-09T03:59:03.944832244Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":14752} +{"time":"2026-08-09T03:59:03.949817464Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":444} +{"time":"2026-08-09T03:59:03.951711505Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":15916} +{"time":"2026-08-09T03:59:03.954435107Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":963} +{"time":"2026-08-09T03:59:03.954578098Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":16937} +{"time":"2026-08-09T03:59:03.955021925Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":119} +{"time":"2026-08-09T03:59:05.531386979Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":1,"events_offset":4,"events_lines":2,"console_offset":4,"console_lines":2} +{"time":"2026-08-09T03:59:06.77622938Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T03:59:20.494664032Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":6,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T03:59:20.585865699Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T03:59:35.494257506Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":8,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T03:59:35.590824829Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T03:59:50.509459052Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":1,"history_lines":1,"events_offset":10,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T03:59:51.514395751Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:00:05.494701492Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":12,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T04:00:05.595719639Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:00:20.494249891Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":14,"events_lines":2,"console_offset":6,"console_lines":2} +{"time":"2026-08-09T04:00:20.621562887Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:00:35.517415448Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":2,"history_lines":1,"events_offset":16,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T04:00:36.625730885Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:00:50.515257082Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":3,"history_lines":1,"events_offset":18,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T04:00:51.502793371Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:01:05.494754271Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":20,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T04:01:05.610106165Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:01:20.494400825Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":22,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T04:01:20.601141197Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:01:35.512242154Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":4,"history_lines":1,"events_offset":24,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T04:01:36.40914958Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:01:50.494503273Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":26,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T04:01:50.59999154Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:02:05.515046749Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":5,"history_lines":1,"events_offset":28,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T04:02:06.39796112Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:02:20.509714476Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":6,"history_lines":1,"events_offset":30,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T04:02:21.325931349Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:02:35.494334742Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":32,"events_lines":2,"console_offset":7,"console_lines":10} +{"time":"2026-08-09T04:02:35.59436135Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:02:50.494250642Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":34,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T04:02:50.613518689Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:03:05.512934467Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":7,"history_lines":1,"events_offset":36,"events_lines":2,"console_offset":16,"console_lines":2} +{"time":"2026-08-09T04:03:06.548287522Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:03:20.494529587Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":38,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T04:03:20.582860961Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:03:35.494527188Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":40,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T04:03:35.607972831Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:03:50.509533438Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":8,"history_lines":1,"events_offset":42,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T04:03:51.431848812Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:04:05.494917144Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":44,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T04:04:05.596754607Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:04:20.507762047Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":9,"history_lines":1,"events_offset":46,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T04:04:21.447073466Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:04:35.509308038Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":10,"history_lines":1,"events_offset":48,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T04:04:36.471928253Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:04:50.494530256Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":50,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T04:04:50.592073322Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:05:05.494375076Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":52,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T04:05:05.604310138Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:05:20.512379195Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":11,"history_lines":1,"events_offset":54,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T04:05:21.491774019Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:05:35.494839434Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":56,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T04:05:35.592901032Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:05:50.49492532Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":58,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T04:05:50.59888656Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:06:05.514147554Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":12,"history_lines":1,"events_offset":60,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T04:06:06.480804234Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:06:20.514491453Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":13,"history_lines":1,"events_offset":62,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T04:06:21.452884273Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:06:35.494286866Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":64,"events_lines":2,"console_offset":18,"console_lines":11} +{"time":"2026-08-09T04:06:35.609203544Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:06:50.494175318Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":66,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T04:06:50.614345309Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:07:05.515682555Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":14,"history_lines":1,"events_offset":68,"events_lines":2,"console_offset":28,"console_lines":2} +{"time":"2026-08-09T04:07:06.427671059Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:07:20.494222615Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":70,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T04:07:20.573884809Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:07:35.507952716Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":15,"history_lines":1,"events_offset":72,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T04:07:36.475332313Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:07:50.494214091Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":74,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T04:07:50.753409176Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:08:05.494655816Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":76,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T04:08:05.586476128Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:08:20.511325739Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":16,"history_lines":1,"events_offset":78,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T04:08:21.410353753Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:08:35.509606509Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":17,"history_lines":1,"events_offset":80,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T04:08:36.420905539Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:08:50.494669295Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":82,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T04:08:50.604423316Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:09:05.494987144Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":84,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T04:09:05.607036127Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:09:20.510454935Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":18,"history_lines":1,"events_offset":86,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T04:09:21.483617947Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:09:35.494837249Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":88,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T04:09:35.610548227Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:09:50.507950635Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":19,"history_lines":1,"events_offset":90,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T04:09:51.511448364Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:10:05.520637653Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":20,"history_lines":1,"events_offset":92,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T04:10:06.438117864Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:10:20.49470281Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":94,"events_lines":2,"console_offset":30,"console_lines":11} +{"time":"2026-08-09T04:10:20.613679557Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:10:35.494489332Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":96,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T04:10:35.60676065Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:10:50.50862038Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":21,"history_lines":1,"events_offset":98,"events_lines":2,"console_offset":40,"console_lines":2} +{"time":"2026-08-09T04:10:51.485985122Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:11:05.494135887Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":100,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T04:11:05.612280741Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:11:20.494686423Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":102,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T04:11:20.598556666Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:11:35.515806086Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":22,"history_lines":1,"events_offset":104,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T04:11:36.484944909Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:11:50.494589134Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":106,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T04:11:50.596223304Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:12:05.513002951Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":23,"history_lines":1,"events_offset":108,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T04:12:06.484107421Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:12:20.494320474Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":110,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T04:12:20.602551441Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:12:35.508125938Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":24,"history_lines":1,"events_offset":112,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T04:12:36.475135091Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:12:50.494789433Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":114,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T04:12:50.602319198Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:13:05.512828195Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":25,"history_lines":1,"events_offset":116,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T04:13:06.512719752Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:13:20.494722423Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":118,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T04:13:20.603843019Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:13:35.494491559Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":120,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T04:13:35.616937106Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:13:50.509869659Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":26,"history_lines":1,"events_offset":122,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T04:13:51.486433037Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:14:05.512616517Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":27,"history_lines":1,"events_offset":124,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T04:14:06.69457877Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:14:20.494738585Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":126,"events_lines":2,"console_offset":42,"console_lines":11} +{"time":"2026-08-09T04:14:20.598086436Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:14:35.494861868Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":128,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T04:14:35.606033152Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:14:50.50832981Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":28,"history_lines":1,"events_offset":130,"events_lines":2,"console_offset":52,"console_lines":2} +{"time":"2026-08-09T04:14:51.451962725Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:15:05.494325431Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":132,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T04:15:05.599151827Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:15:20.512939886Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":29,"history_lines":1,"events_offset":134,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T04:15:21.39376644Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:15:35.494565716Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":136,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T04:15:35.597861377Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:15:50.494985286Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":138,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T04:15:50.61366858Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:16:05.509505556Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":30,"history_lines":1,"events_offset":140,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T04:16:06.464859955Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:16:20.512387578Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":31,"history_lines":1,"events_offset":142,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T04:16:21.456484085Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:16:35.494311848Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":144,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T04:16:35.614320532Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:16:50.494123741Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":146,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T04:16:50.597183977Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:17:05.507650032Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":32,"history_lines":1,"events_offset":148,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T04:17:06.496156926Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:17:20.494465778Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":150,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T04:17:20.608687507Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:17:35.494337018Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":152,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T04:17:35.598855029Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:17:50.511814815Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":33,"history_lines":1,"events_offset":154,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T04:17:51.379590142Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:18:05.507926178Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":34,"history_lines":1,"events_offset":156,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T04:18:06.425217866Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:18:20.494329994Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":158,"events_lines":2,"console_offset":54,"console_lines":13} +{"time":"2026-08-09T04:18:20.597550528Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:18:35.494357619Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":160,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T04:18:35.590558781Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:18:50.526101792Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":35,"history_lines":1,"events_offset":162,"events_lines":2,"console_offset":66,"console_lines":2} +{"time":"2026-08-09T04:18:51.756248392Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:19:05.494443616Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":164,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T04:19:05.600417147Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:19:20.509453223Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":36,"history_lines":1,"events_offset":166,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T04:19:21.44743209Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:19:35.494186649Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":168,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T04:19:35.598016278Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:19:50.494255354Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":170,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T04:19:50.607645062Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:20:05.512253299Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":37,"history_lines":1,"events_offset":172,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T04:20:06.424682607Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:20:20.514248726Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":38,"history_lines":1,"events_offset":174,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T04:20:21.478103921Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:20:35.494263985Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":176,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T04:20:35.622961052Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:20:50.494962002Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":178,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T04:20:50.611850033Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:21:05.517835067Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":39,"history_lines":1,"events_offset":180,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T04:21:06.404943823Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:21:20.494738486Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":182,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T04:21:20.602524281Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:21:35.508789164Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":40,"history_lines":1,"events_offset":184,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T04:21:36.443085607Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:21:50.511158411Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":41,"history_lines":1,"events_offset":186,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T04:21:51.460178906Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:22:05.494482305Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":188,"events_lines":2,"console_offset":68,"console_lines":11} +{"time":"2026-08-09T04:22:05.586362541Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:22:20.494728352Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":190,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T04:22:20.599759017Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:22:35.513754334Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":42,"history_lines":1,"events_offset":192,"events_lines":2,"console_offset":78,"console_lines":2} +{"time":"2026-08-09T04:22:36.516635185Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:22:50.494586997Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":194,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T04:22:50.571528834Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:23:05.494079314Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":196,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T04:23:05.598391284Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:23:20.514134103Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":43,"history_lines":1,"events_offset":198,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T04:23:21.457824889Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:23:35.494479121Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":200,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T04:23:35.592895767Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:23:50.518950272Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":44,"history_lines":1,"events_offset":202,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T04:23:51.404637571Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:24:05.494720966Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":204,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T04:24:05.600554603Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:24:20.51001913Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":45,"history_lines":1,"events_offset":206,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T04:24:21.455271551Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:24:35.49440308Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":208,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T04:24:35.587198082Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:24:50.512521872Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":46,"history_lines":1,"events_offset":210,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T04:24:51.470179091Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:25:05.494790615Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":212,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T04:25:05.614937066Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:25:20.494768189Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":214,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T04:25:20.603889882Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:25:35.510484888Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":47,"history_lines":1,"events_offset":216,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T04:25:36.442214296Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:25:50.512621688Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":48,"history_lines":1,"events_offset":218,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T04:25:51.447687898Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:26:05.494610601Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":220,"events_lines":2,"console_offset":80,"console_lines":11} +{"time":"2026-08-09T04:26:05.591575531Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:26:20.49440059Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":222,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T04:26:20.578020301Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:26:35.508448912Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":49,"history_lines":1,"events_offset":224,"events_lines":2,"console_offset":90,"console_lines":2} +{"time":"2026-08-09T04:26:36.403987032Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:26:50.494361014Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":226,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T04:26:50.602435073Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:27:05.507911257Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":50,"history_lines":1,"events_offset":228,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T04:27:06.445082363Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:27:20.494195074Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":230,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T04:27:20.614605046Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:27:35.494248784Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":232,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T04:27:35.607893177Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:27:50.514285947Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":51,"history_lines":1,"events_offset":234,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T04:27:51.395463717Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:28:05.514892856Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":52,"history_lines":1,"events_offset":236,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T04:28:06.423194446Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:28:20.49418678Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":238,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T04:28:20.597735182Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:28:35.494119279Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":240,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T04:28:35.592493251Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:28:50.5094627Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":53,"history_lines":1,"events_offset":242,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T04:28:51.45483569Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:29:05.494391144Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":244,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T04:29:05.591119154Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:29:20.494298466Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":246,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T04:29:20.596120363Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:29:35.513763328Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":54,"history_lines":1,"events_offset":248,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T04:29:36.465513234Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:29:50.513046942Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":55,"history_lines":1,"events_offset":250,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T04:29:51.472667426Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:30:05.494740686Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":252,"events_lines":2,"console_offset":92,"console_lines":11} +{"time":"2026-08-09T04:30:05.586971324Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:30:20.515798418Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":56,"history_lines":1,"events_offset":254,"events_lines":2,"console_offset":102,"console_lines":2} +{"time":"2026-08-09T04:30:21.494651995Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:30:35.494231232Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":256,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T04:30:35.58729982Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:30:50.494976906Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":258,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T04:30:50.601902681Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:31:05.510769779Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":57,"history_lines":1,"events_offset":260,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T04:31:06.780111879Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:31:20.49453644Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":262,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T04:31:20.599562246Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:31:35.510004044Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":58,"history_lines":1,"events_offset":264,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T04:31:36.458139374Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:31:50.494509411Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":266,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T04:31:50.575654427Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:32:05.512223467Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":59,"history_lines":1,"events_offset":268,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T04:32:06.469562043Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:32:20.494205295Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":270,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T04:32:20.584293928Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:32:35.509793195Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":60,"history_lines":1,"events_offset":272,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T04:32:36.535545937Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:32:50.494041072Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":274,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T04:32:50.594918103Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:33:05.49403774Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":276,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T04:33:05.607036499Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:33:20.511846379Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":61,"history_lines":1,"events_offset":278,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T04:33:21.48470859Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:33:35.515062737Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":62,"history_lines":1,"events_offset":280,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T04:33:36.506124978Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:33:50.4946323Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":282,"events_lines":2,"console_offset":104,"console_lines":11} +{"time":"2026-08-09T04:33:50.612161047Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:34:05.494640716Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":284,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T04:34:05.587261158Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:34:20.507829114Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":63,"history_lines":1,"events_offset":286,"events_lines":2,"console_offset":114,"console_lines":2} +{"time":"2026-08-09T04:34:21.464551249Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:34:35.494345284Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":288,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T04:34:35.606869158Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:34:50.494397686Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":290,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T04:34:50.599232901Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:35:05.508409123Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":64,"history_lines":1,"events_offset":292,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T04:35:06.433911202Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:35:20.494170669Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":294,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T04:35:20.613977234Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:35:35.509564094Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":65,"history_lines":1,"events_offset":296,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T04:35:36.475546977Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:35:50.508108638Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":66,"history_lines":1,"events_offset":298,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T04:35:51.45083907Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:36:05.494407838Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":300,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T04:36:05.588011312Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:36:20.494349239Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":302,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T04:36:20.614846145Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:36:35.510501251Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":67,"history_lines":1,"events_offset":304,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T04:36:36.571048121Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:36:50.494472539Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":306,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T04:36:50.609659468Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:37:05.494605434Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":308,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T04:37:05.590686834Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:37:20.510275452Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":68,"history_lines":1,"events_offset":310,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T04:37:21.512921238Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:37:35.50955353Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":69,"history_lines":1,"events_offset":312,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T04:37:36.500587981Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:37:50.494388687Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":314,"events_lines":2,"console_offset":116,"console_lines":13} +{"time":"2026-08-09T04:37:50.598720388Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:38:05.494193752Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":316,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T04:38:05.638784236Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:38:20.530757676Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":70,"history_lines":1,"events_offset":318,"events_lines":2,"console_offset":128,"console_lines":2} +{"time":"2026-08-09T04:38:21.632845005Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:38:35.494438726Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":320,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T04:38:35.594398657Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:38:50.514389068Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":71,"history_lines":1,"events_offset":322,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T04:38:51.399815543Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:39:05.494294314Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":324,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T04:39:05.605503032Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:39:20.494680808Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":326,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T04:39:20.611296788Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:39:35.510121829Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":72,"history_lines":1,"events_offset":328,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T04:39:36.465679131Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:39:50.512824888Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":73,"history_lines":1,"events_offset":330,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T04:39:51.478305625Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:40:05.494062759Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":332,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T04:40:05.580589634Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:40:20.494165168Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":334,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T04:40:20.607065019Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:40:35.512178504Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":74,"history_lines":1,"events_offset":336,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T04:40:36.602134021Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:40:50.494375721Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":338,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T04:40:50.583296822Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:41:05.518526338Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":75,"history_lines":1,"events_offset":340,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T04:41:06.435876732Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:41:20.494762288Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":342,"events_lines":2,"console_offset":130,"console_lines":6} +{"time":"2026-08-09T04:41:20.603430974Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:41:35.514811938Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":76,"history_lines":1,"events_offset":344,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T04:41:36.473044784Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:41:50.494076612Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":346,"events_lines":2,"console_offset":131,"console_lines":1} +{"time":"2026-08-09T04:41:50.601570222Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:42:05.511952413Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":77,"history_lines":1,"events_offset":348,"events_lines":2,"console_offset":136,"console_lines":6} +{"time":"2026-08-09T04:42:06.491073426Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:42:20.494758077Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":350,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T04:42:20.606830814Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:42:35.494254813Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":352,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T04:42:35.595459447Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:42:50.509597712Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":78,"history_lines":1,"events_offset":354,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T04:42:51.385361692Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:43:05.494296853Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":356,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T04:43:05.584770185Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:43:20.494194739Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":358,"events_lines":2,"console_offset":142,"console_lines":2} +{"time":"2026-08-09T04:43:20.605028176Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:43:35.511553269Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":79,"history_lines":1,"events_offset":360,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T04:43:36.422803214Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:43:50.51209143Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":80,"history_lines":1,"events_offset":362,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T04:43:51.371685359Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:44:05.494414492Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":364,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T04:44:05.591512538Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:44:20.494476569Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":366,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T04:44:20.61786034Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:44:35.494661644Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":368,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T04:44:35.600856267Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:44:50.512310947Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":81,"history_lines":1,"events_offset":370,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T04:44:51.463291334Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:45:05.494886009Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":372,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T04:45:05.584389307Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:45:20.494304778Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":374,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T04:45:20.594809211Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:45:35.510247414Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":82,"history_lines":1,"events_offset":376,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T04:45:36.477447729Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:45:50.515143514Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":83,"history_lines":1,"events_offset":378,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T04:45:51.542141686Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:46:05.494993318Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":380,"events_lines":2,"console_offset":143,"console_lines":10} +{"time":"2026-08-09T04:46:05.60243871Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:46:20.494284253Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":382,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:46:20.608258694Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:46:35.494646706Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":384,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:46:35.598907205Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:46:50.514394328Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":84,"history_lines":1,"events_offset":386,"events_lines":2,"console_offset":152,"console_lines":2} +{"time":"2026-08-09T04:46:51.494084865Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:47:05.494697595Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":388,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:47:05.586587514Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:47:20.494438098Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":390,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:47:20.597334702Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:47:35.521023173Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":85,"history_lines":1,"events_offset":392,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:47:36.445223277Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:47:50.494788572Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":394,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:47:50.590619221Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:48:05.49423104Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":396,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:48:05.600318316Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:48:20.509807197Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":86,"history_lines":1,"events_offset":398,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:48:21.489740194Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:48:35.494208141Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":400,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:48:35.591006311Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:48:50.515494589Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":87,"history_lines":1,"events_offset":402,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:48:51.427014966Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:49:05.494430456Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":404,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:49:05.619853108Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:49:20.494243574Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":406,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:49:20.601941856Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:49:35.513932029Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":88,"history_lines":1,"events_offset":408,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:49:36.457240216Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:49:50.494940016Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":410,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:49:50.581649644Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:50:05.494720247Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":412,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:50:05.608214779Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:50:20.494063881Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":414,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:50:20.603378552Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:50:35.519647341Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":89,"history_lines":1,"events_offset":416,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:50:36.503626552Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:50:50.51423598Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":90,"history_lines":1,"events_offset":418,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T04:50:51.459574025Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:51:05.494376593Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":420,"events_lines":2,"console_offset":154,"console_lines":11} +{"time":"2026-08-09T04:51:05.591793843Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:51:20.494529619Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":422,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:51:20.639974097Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:51:35.494240177Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":424,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:51:35.59915761Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:51:50.508256175Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":91,"history_lines":1,"events_offset":426,"events_lines":2,"console_offset":164,"console_lines":2} +{"time":"2026-08-09T04:51:51.432054336Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:52:05.494231656Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":428,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:52:05.590635379Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:52:20.494183843Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":430,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:52:20.609362605Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:52:35.512240353Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":92,"history_lines":1,"events_offset":432,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:52:36.547000959Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:52:50.494409211Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":434,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:52:50.611077694Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:53:05.494665223Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":436,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:53:05.604913862Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:53:20.513692859Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":93,"history_lines":1,"events_offset":438,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:53:21.370170293Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:53:35.494265173Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":440,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:53:35.588156095Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:53:50.515309054Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":94,"history_lines":1,"events_offset":442,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:53:51.508264221Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:54:05.494267129Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":444,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:54:05.598594866Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:54:20.494465833Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":446,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:54:20.599852885Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:54:35.512705924Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":95,"history_lines":1,"events_offset":448,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:54:36.45915457Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:54:50.494329668Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":450,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:54:50.606394944Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:55:05.494173382Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":452,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:55:05.606685612Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:55:20.494519313Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":454,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:55:20.600694415Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:55:35.509767756Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":96,"history_lines":1,"events_offset":456,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:55:36.518053017Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:55:50.511111663Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":97,"history_lines":1,"events_offset":458,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T04:55:51.460606406Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:56:05.494354453Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":460,"events_lines":2,"console_offset":166,"console_lines":11} +{"time":"2026-08-09T04:56:05.588540788Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:56:20.49440425Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":462,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T04:56:20.595050994Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:56:35.494373973Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":464,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T04:56:35.609327704Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:56:50.51145138Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":98,"history_lines":1,"events_offset":466,"events_lines":2,"console_offset":176,"console_lines":2} +{"time":"2026-08-09T04:56:51.609244462Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:57:05.494213007Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":468,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T04:57:05.598515761Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:57:20.494450349Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":470,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T04:57:20.596086948Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:57:35.512624981Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":99,"history_lines":1,"events_offset":472,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T04:57:36.509474508Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:57:50.494590228Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":474,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T04:57:50.582329386Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:58:05.494166836Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":476,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T04:58:05.584219865Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:58:20.512387587Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":100,"history_lines":1,"events_offset":478,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T04:58:21.442742736Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:58:35.494461513Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":480,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T04:58:35.704810548Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:58:50.511948147Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":101,"history_lines":1,"events_offset":482,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T04:58:51.416645299Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:59:05.494670676Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":484,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T04:59:05.594712461Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:59:20.494633402Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":486,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T04:59:20.592347073Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:59:35.510861324Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":102,"history_lines":1,"events_offset":488,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T04:59:36.435711548Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T04:59:50.494866252Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":490,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T04:59:50.60422558Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:00:05.510416009Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":103,"history_lines":1,"events_offset":492,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:00:06.405297236Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:00:20.512054289Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":104,"history_lines":2,"events_offset":494,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:00:21.415448411Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:00:35.494696267Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":496,"events_lines":2,"console_offset":178,"console_lines":13} +{"time":"2026-08-09T05:00:35.595572837Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:00:39.446368581Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"} +{"time":"2026-08-09T05:00:39.464038823Z","level":"INFO","msg":"filestream: sending request","total_files":3,"history_offset":106,"history_lines":1,"console_offset":190,"console_lines":31,"uploaded_len":3,"complete":true,"exit_code":0} +{"time":"2026-08-09T05:00:40.496960895Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:00:40.498370506Z","level":"INFO","msg":"handler: operation stats","stats":{}} +{"time":"2026-08-09T05:00:40.552225292Z","level":"INFO","msg":"stream: finishing up"} +{"time":"2026-08-09T05:00:40.552256718Z","level":"INFO","msg":"handler: closed"} +{"time":"2026-08-09T05:00:40.552367204Z","level":"INFO","msg":"sender: closed"} +{"time":"2026-08-09T05:00:40.552371118Z","level":"INFO","msg":"stream: all finished"} diff --git a/wandb/run-20260809_035819-cvzjg5ej/logs/debug.log b/wandb/run-20260809_035819-cvzjg5ej/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..eb4b718d5f0d6fdd116419837f4665b306461465 --- /dev/null +++ b/wandb/run-20260809_035819-cvzjg5ej/logs/debug.log @@ -0,0 +1,28 @@ +2026-08-09 03:58:19,281 INFO MainThread:2921752 [wandb_setup.py:_flush():81] Current SDK version is 0.28.1 +2026-08-09 03:58:19,281 INFO MainThread:2921752 [wandb_setup.py:_flush():81] Configure stats pid to 2921752 +2026-08-09 03:58:19,281 INFO MainThread:2921752 [wandb_setup.py:_flush():81] Loading settings from environment variables +2026-08-09 03:58:19,281 INFO MainThread:2921752 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_035819-cvzjg5ej/logs/debug.log +2026-08-09 03:58:19,281 INFO MainThread:2921752 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_035819-cvzjg5ej/logs/debug-internal.log +2026-08-09 03:58:19,281 INFO MainThread:2921752 [wandb_init.py:init():772] calling init triggers +2026-08-09 03:58:19,281 INFO MainThread:2921752 [wandb_init.py:init():777] wandb.init called with sweep_config: {} +config: {'_wandb': {}} +2026-08-09 03:58:19,281 INFO MainThread:2921752 [wandb_init.py:init():820] starting backend +2026-08-09 03:58:19,281 INFO MainThread:2921752 [wandb_init.py:init():826] Connected to an existing wandb-core service via WANDB_SERVICE +2026-08-09 03:58:19,281 INFO MainThread:2921752 [wandb_init.py:init():835] sending inform_init request +2026-08-09 03:58:19,544 INFO MainThread:2921752 [wandb_init.py:init():840] backend started and connected +2026-08-09 03:58:19,547 INFO MainThread:2921752 [wandb_init.py:init():910] updated telemetry +2026-08-09 03:58:19,554 INFO MainThread:2921752 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout +2026-08-09 03:58:19,820 INFO MainThread:2921752 [wandb_init.py:init():978] starting run threads in backend +2026-08-09 03:58:19,892 INFO MainThread:2921752 [wandb_run.py:_console_start():2621] atexit reg +2026-08-09 03:58:19,892 INFO MainThread:2921752 [wandb_run.py:_redirect():2471] redirect: wrap_raw +2026-08-09 03:58:19,892 INFO MainThread:2921752 [wandb_run.py:_redirect():2540] Wrapping output streams. +2026-08-09 03:58:19,892 INFO MainThread:2921752 [wandb_run.py:_redirect():2563] Redirects installed. +2026-08-09 03:58:19,895 INFO MainThread:2921752 [wandb_init.py:init():1016] run started, returning control to user process +2026-08-09 03:58:19,896 INFO MainThread:2921752 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.15.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 94, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 's10', 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/mlp-s10-94L_run', 'per_device_train_batch_size': 128, 'num_train_epochs': 1, 'max_steps': 1500, 'learning_rate': 0.001, 'lr_scheduler_type': 'constant', 'lr_scheduler_kwargs': None, 'warmup_steps': 0, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.01, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 4, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-s10-94L-15.9M-20260809-035818', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 50, 'eval_delay': 0, 'per_device_eval_batch_size': 128, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': True, 'hub_token': '', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/A-mlp-s10-94L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1} +2026-08-09 03:58:19,899 INFO MainThread:2921752 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 15949440 - > +2026-08-09 03:58:19,899 INFO MainThread:2921752 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 15949440 None +2026-08-09 05:00:38,675 INFO MainThread:2921752 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/huggingface/cvzjg5ej +2026-08-09 05:00:38,675 INFO MainThread:2921752 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0 +2026-08-09 05:00:38,676 INFO MainThread:2921752 [wandb_run.py:_restore():2570] restore +2026-08-09 05:00:38,676 INFO MainThread:2921752 [wandb_run.py:_restore():2576] restore done +2026-08-09 05:00:40,551 INFO MainThread:2921752 [wandb_run.py:_footer_sync_info():3993] logging synced files diff --git a/wandb/run-20260809_035819-cvzjg5ej/run-cvzjg5ej.wandb b/wandb/run-20260809_035819-cvzjg5ej/run-cvzjg5ej.wandb new file mode 100644 index 0000000000000000000000000000000000000000..5a0e2c083fc486dc44bd50503336fa06a4230637 --- /dev/null +++ b/wandb/run-20260809_035819-cvzjg5ej/run-cvzjg5ej.wandb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:42e28ca89c157db3fd0e1b8d55b132aed26283c09abd2a439124929bd964170e +size 17095417 diff --git a/wandb/run-20260809_050050-59pftr14/files/config.yaml b/wandb/run-20260809_050050-59pftr14/files/config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..077b5ea72bd63662485ccf863fab79ed5cfe171f --- /dev/null +++ b/wandb/run-20260809_050050-59pftr14/files/config.yaml @@ -0,0 +1,433 @@ +_name_or_path: + value: "" +_wandb: + value: + cli_version: 0.28.1 + e: + ou0ddnt36zj23smd7g76a30q6z1cg54s: + args: + - --config + - configs/baseline.yaml + - --variants + - glu-linear-94L + - --push + codePath: sweep.py + codePathLocal: sweep.py + cpu_count: 112 + cpu_count_logical: 224 + cudaVersion: "12.4" + disk: + /: + total: "1560765693952" + used: "708235583488" + email: deepnevro@gmail.com + executable: /mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python + git: + commit: 34b8d2e8f9a0c5751333310e69fa0c1056381deb + remote: https://github.com/deepnevro/Activation.git + gpu: NVIDIA H100 80GB HBM3 + gpu_count: 8 + gpu_nvidia: + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-39c684a5-fde6-83d7-1663-0859795881ae + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-132944c4-b689-2b5f-89a4-d730401677ab + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bc6c3e3c-9b90-09ca-c034-774961847c54 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea + host: deeplens-k3s-node1 + memory: + total: "2164089937920" + os: Linux-5.15.0-126-generic-x86_64-with-glibc2.35 + program: /mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py + python: CPython 3.11.15 + root: /mnt/data/zainulabideen/zain-exp/notebooks/Activation + startedAt: "2026-08-09T05:00:50.313572Z" + writerId: ou0ddnt36zj23smd7g76a30q6z1cg54s + m: + - "1": train/global_step + "6": + - 3 + "7": [] + - "2": '*' + "5": 1 + "6": + - 1 + "7": [] + python_version: 3.11.15 + t: + "1": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "2": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "3": + - 2 + - 7 + - 13 + - 19 + - 41 + - 62 + - 66 + "4": 3.11.15 + "5": 0.28.1 + "6": 5.15.0.dev0 + "9": + "1": transformers_trainer + "12": 0.28.1 + "13": linux-x86_64 +accelerator_config: + value: + dispatch_batches: null + even_batches: true + gradient_accumulation_kwargs: null + non_blocking: false + split_batches: false + use_seedable_sampler: true +activation: + value: linear +adam_beta1: + value: 0.9 +adam_beta2: + value: 0.999 +adam_epsilon: + value: 1e-08 +architectures: + value: null +attention_bias: + value: false +attention_dropout: + value: 0 +auto_find_batch_size: + value: false +average_tokens_across_devices: + value: true +batch_eval_metrics: + value: false +bf16: + value: true +bf16_full_eval: + value: false +bos_token_id: + value: 1 +chunk_size_feed_forward: + value: 0 +data_seed: + value: 42 +dataloader_drop_last: + value: false +dataloader_in_order: + value: true +dataloader_multiprocessing_context: + value: null +dataloader_num_workers: + value: 0 +dataloader_persistent_workers: + value: false +dataloader_pin_memory: + value: true +dataloader_prefetch_factor: + value: null +ddp_backend: + value: null +ddp_broadcast_buffers: + value: null +ddp_bucket_cap_mb: + value: null +ddp_find_unused_parameters: + value: null +ddp_static_graph: + value: null +ddp_timeout: + value: 1800 +debug: + value: [] +deepspeed: + value: null +disable_tqdm: + value: false +do_eval: + value: true +do_predict: + value: false +do_train: + value: false +dtype: + value: null +enable_jit_checkpoint: + value: false +eos_token_id: + value: 2 +eval_accumulation_steps: + value: null +eval_delay: + value: 0 +eval_do_concat_batches: + value: true +eval_on_start: + value: false +eval_steps: + value: 50 +eval_strategy: + value: steps +eval_use_gather_object: + value: false +fp16: + value: false +fp16_full_eval: + value: false +fsdp: + value: null +fsdp_config: + value: null +full_determinism: + value: false +gradient_accumulation_steps: + value: 4 +gradient_checkpointing: + value: false +gradient_checkpointing_kwargs: + value: null +greater_is_better: + value: null +head_dim: + value: 32 +hidden_act: + value: silu +hidden_size: + value: 128 +hub_always_push: + value: false +hub_model_id: + value: w-ahmad/A-glu-linear-94L +hub_private_repo: + value: null +hub_revision: + value: null +hub_strategy: + value: every_save +hub_token: + value: +id2label: + value: + "0": LABEL_0 + "1": LABEL_1 +ignore_data_skip: + value: false +include_for_metrics: + value: [] +include_num_input_tokens_seen: + value: "no" +initializer_range: + value: 0.02 +intermediate_size: + value: 256 +is_encoder_decoder: + value: false +label_names: + value: null +label_smoothing_factor: + value: 0 +label2id: + value: + LABEL_0: 0 + LABEL_1: 1 +learning_rate: + value: 0.001 +length_column_name: + value: length +liger_kernel_config: + value: null +load_best_model_at_end: + value: false +local_rank: + value: -1 +log_level: + value: passive +log_level_replica: + value: warning +log_on_each_node: + value: true +logging_first_step: + value: false +logging_nan_inf_filter: + value: true +logging_steps: + value: 20 +logging_strategy: + value: steps +lr_scheduler_kwargs: + value: null +lr_scheduler_type: + value: constant +max_grad_norm: + value: 1 +max_position_embeddings: + value: 512 +max_steps: + value: 1500 +metric_for_best_model: + value: null +mlp_bias: + value: false +mlp_type: + value: glu +model/num_parameters: + value: 15949440 +model_type: + value: tiny_llama +neftune_noise_alpha: + value: null +num_attention_heads: + value: 4 +num_hidden_layers: + value: 94 +num_key_value_heads: + value: 4 +num_train_epochs: + value: 1 +optim: + value: adamw_torch_fused +optim_args: + value: null +optim_target_modules: + value: null +output_attentions: + value: false +output_dir: + value: out/glu-linear-94L_run +output_hidden_states: + value: false +pad_token_id: + value: 0 +parallelism_config: + value: null +per_device_eval_batch_size: + value: 128 +per_device_train_batch_size: + value: 128 +prediction_loss_only: + value: false +pretraining_tp: + value: 1 +problem_type: + value: null +project: + value: huggingface +push_to_hub: + value: true +remove_unused_columns: + value: false +report_to: + value: + - wandb +restore_callback_states_from_checkpoint: + value: false +resume_from_checkpoint: + value: null +return_dict: + value: true +rms_norm_eps: + value: 1e-06 +rope_parameters: + value: + rope_theta: 10000 + rope_type: default +run_name: + value: LM-glu-linear-94L-15.9M-20260809-050049 +save_on_each_node: + value: false +save_only_model: + value: false +save_steps: + value: 100 +save_strategy: + value: steps +save_total_limit: + value: null +seed: + value: 42 +skip_memory_metrics: + value: true +tf32: + value: null +tie_word_embeddings: + value: true +tokenizer_name: + value: w-ahmad/tiny-stories-tokenizer +torch_compile: + value: false +torch_compile_backend: + value: null +torch_compile_mode: + value: null +torch_empty_cache_steps: + value: null +trackio_bucket_id: + value: null +trackio_space_id: + value: null +trackio_static_space_id: + value: null +train_sampling_strategy: + value: random +transformers_version: + value: 5.15.0.dev0 +use_cache: + value: false +use_cpu: + value: false +use_liger_kernel: + value: false +vocab_size: + value: 4096 +warmup_steps: + value: 0 +weight_decay: + value: 0.01 diff --git a/wandb/run-20260809_050050-59pftr14/files/output.log b/wandb/run-20260809_050050-59pftr14/files/output.log new file mode 100644 index 0000000000000000000000000000000000000000..e8f2f7850511ce1ea7e0f0c8ce0a3d6eccb722e9 --- /dev/null +++ b/wandb/run-20260809_050050-59pftr14/files/output.log @@ -0,0 +1,221 @@ +[transformers] `use_return_dict` is deprecated! Use `return_dict` instead! +[INFO] Causal mask (float with -inf) applied to all attention layers. +/mnt/data/zainulabideen/zain-exp/notebooks/Activation/exp.py:487: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1831.) + "std": tensor.std().item(), + 7%|██▌ | 100/1500 [03:44<43:52, 1.88s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '28.45', 'grad_norm': '3.625', 'learning_rate': '0.001', 'epoch': '0.01078', 'train/total_time_seconds': '33.14', 'train/time_per_step_avg': '1.657', 'train/epoch_time_elapsed': '41.52', 'train/estimated_remaining_minutes': '40.88', 'train/global/act/norm': '8.914e+04', 'train/global/act/mean': '-0.002708', 'train/global/act/std': '0.4187', 'train/global/act/max_abs': '8.323', 'train/global/act/frac_near_dtype_limit': '0', 'train/global/act/frac_near_user_limit': '0', 'train/global/grad/norm': '3.6', 'train/global/grad/mean': '-7.779e-08', 'train/global/grad/std': '0.0004509', 'train/global/grad/max_abs': '0.1279', 'train/global/grad/frac_near_dtype_limit': '0', 'train/global/grad/frac_near_user_limit': '0', 'train/global/param/norm': '174.8', 'train/global/param/mean': '0.001521', 'train/global/param/std': '0.04376', 'train/global/param/max_abs': '1', 'train/global/param/frac_near_dtype_limit': '0', 'train/global/param/frac_near_user_limit': '0', 'train/layer__model_layers_49/param/norm': '17.93', 'train/layer__model_layers_49/param/mean': '0.001503', 'train/layer__model_layers_49/param/std': '0.04425', 'train/layer__model_layers_49/param/max_abs': '1', 'train/layer__model_layers_49/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_49/param/frac_near_user_limit': '0', 'train/layer__model_layers_7/param/norm': '17.94', 'train/layer__model_layers_7/param/mean': '0.00151', 'train/layer__model_layers_7/param/std': '0.04425', 'train/layer__model_layers_7/param/max_abs': '1', 'train/layer__model_layers_7/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_7/param/frac_near_user_limit': '0', 'train/layer__model_layers_86/param/norm': '17.93', 'train/layer__model_layers_86/param/mean': '0.00157', 'train/layer__model_layers_86/param/std': '0.04425', 'train/layer__model_layers_86/param/max_abs': '1', 'train/layer__model_layers_86/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_86/param/frac_near_user_limit': '0', 'train/layer_model_layers_4/act/norm': '8928', 'train/layer_model_layers_4/act/mean': '-0.007839', 'train/layer_model_layers_4/act/std': '0.4122', 'train/layer_model_layers_4/act/max_abs': '5.469', 'train/layer_model_layers_4/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_4/act/frac_near_user_limit': '0', 'train/layer_model_layers_4/grad/norm': '0.7242', 'train/layer_model_layers_4/grad/mean': '-1.734e-06', 'train/layer_model_layers_4/grad/std': '0.0008939', 'train/layer_model_layers_4/grad/max_abs': '0.01489', 'train/layer_model_layers_4/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_4/grad/frac_near_user_limit': '0', 'train/layer_model_layers_40/act/norm': '9081', 'train/layer_model_layers_40/act/mean': '0.000482', 'train/layer_model_layers_40/act/std': '0.4191', 'train/layer_model_layers_40/act/max_abs': '4.875', 'train/layer_model_layers_40/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_40/act/frac_near_user_limit': '0', 'train/layer_model_layers_40/grad/norm': '0.1891', 'train/layer_model_layers_40/grad/mean': '-4.847e-07', 'train/layer_model_layers_40/grad/std': '0.0002335', 'train/layer_model_layers_40/grad/max_abs': '0.006805', 'train/layer_model_layers_40/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_40/grad/frac_near_user_limit': '0', 'train/layer_model_layers_34/act/norm': '9060', 'train/layer_model_layers_34/act/mean': '-0.002399', 'train/layer_model_layers_34/act/std': '0.4181', 'train/layer_model_layers_34/act/max_abs': '5.188', 'train/layer_model_layers_34/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_34/act/frac_near_user_limit': '0', 'train/layer_model_layers_34/grad/norm': '0.229', 'train/layer_model_layers_34/grad/mean': '1.348e-07', 'train/layer_model_layers_34/grad/std': '0.0002827', 'train/layer_model_layers_34/grad/max_abs': '0.007446', 'train/layer_model_layers_34/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_34/grad/frac_near_user_limit': '0', 'train/layer_model_layers_27/act/norm': '9016', 'train/layer_model_layers_27/act/mean': '-0.006148', 'train/layer_model_layers_27/act/ +{'loss': '24.25', 'grad_norm': '0.2471', 'learning_rate': '0.001', 'epoch': '0.02157', 'train/total_time_seconds': '62.96', 'train/time_per_step_avg': '1.574', 'train/epoch_time_elapsed': '79.29', 'train/estimated_remaining_minutes': '38.3'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '5.901', 'eval_runtime': '15.77', 'eval_samples_per_second': '604.3', 'eval_steps_per_second': '4.757', 'epoch': '0.02696', 'train/total_time_seconds': '77.87', 'train/time_per_step_avg': '1.557', 'train/epoch_time_elapsed': '113.9', 'train/estimated_remaining_minutes': '37.63'} +{'loss': '23.61', 'grad_norm': '1.281', 'learning_rate': '0.001', 'epoch': '0.03235', 'train/total_time_seconds': '92.78', 'train/time_per_step_avg': '1.546', 'train/epoch_time_elapsed': '132.7', 'train/estimated_remaining_minutes': '37.11'} +{'loss': '22.99', 'grad_norm': '1.477', 'learning_rate': '0.001', 'epoch': '0.04314', 'train/total_time_seconds': '122.6', 'train/time_per_step_avg': '1.532', 'train/epoch_time_elapsed': '170.2', 'train/estimated_remaining_minutes': '36.26'} +{'loss': '22.73', 'grad_norm': '1.641', 'learning_rate': '0.001', 'epoch': '0.05392', 'train/total_time_seconds': '152.4', 'train/time_per_step_avg': '1.524', 'train/epoch_time_elapsed': '207.7', 'train/estimated_remaining_minutes': '35.56'} +{'eval_loss': '5.594', 'eval_runtime': '15.83', 'eval_samples_per_second': '601.7', 'eval_steps_per_second': '4.737', 'epoch': '0.05392', 'train/total_time_seconds': '152.4', 'train/time_per_step_avg': '1.524', 'train/epoch_time_elapsed': '223.6', 'train/estimated_remaining_minutes': '35.56'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 14.58it/s] + 13%|█████▏ | 200/1500 [07:24<40:31, 1.87s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '21.94', 'grad_norm': '2.391', 'learning_rate': '0.001', 'epoch': '0.06471', 'train/total_time_seconds': '182.3', 'train/time_per_step_avg': '1.491', 'train/epoch_time_elapsed': '261.6', 'train/estimated_remaining_minutes': '34.94'} +{'loss': '21.14', 'grad_norm': '1.93', 'learning_rate': '0.001', 'epoch': '0.07549', 'train/total_time_seconds': '212.4', 'train/time_per_step_avg': '1.494', 'train/epoch_time_elapsed': '299.4', 'train/estimated_remaining_minutes': '34.38'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '5.011', 'eval_runtime': '15.69', 'eval_samples_per_second': '607.3', 'eval_steps_per_second': '4.781', 'epoch': '0.08088', 'train/total_time_seconds': '227.3', 'train/time_per_step_avg': '1.494', 'train/epoch_time_elapsed': '333.9', 'train/estimated_remaining_minutes': '34.09'} +{'loss': '20.02', 'grad_norm': '2.234', 'learning_rate': '0.001', 'epoch': '0.08628', 'train/total_time_seconds': '242.2', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '352.9', 'train/estimated_remaining_minutes': '33.81'} +{'loss': '18.95', 'grad_norm': '5.344', 'learning_rate': '0.001', 'epoch': '0.09706', 'train/total_time_seconds': '272.1', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '390.4', 'train/estimated_remaining_minutes': '33.26'} +{'loss': '18.17', 'grad_norm': '2.844', 'learning_rate': '0.001', 'epoch': '0.1078', 'train/total_time_seconds': '302', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '427.9', 'train/estimated_remaining_minutes': '32.71'} +{'eval_loss': '4.437', 'eval_runtime': '15.72', 'eval_samples_per_second': '606.2', 'eval_steps_per_second': '4.772', 'epoch': '0.1078', 'train/total_time_seconds': '302', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '443.7', 'train/estimated_remaining_minutes': '32.71'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 15.25it/s] + 20%|███████▊ | 300/1500 [11:03<37:25, 1.87s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '17.4', 'grad_norm': '1.891', 'learning_rate': '0.001', 'epoch': '0.1186', 'train/total_time_seconds': '331.8', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '481.4', 'train/estimated_remaining_minutes': '32.18'} +{'loss': '16.68', 'grad_norm': '3.359', 'learning_rate': '0.001', 'epoch': '0.1294', 'train/total_time_seconds': '361.7', 'train/time_per_step_avg': '1.493', 'train/epoch_time_elapsed': '519', 'train/estimated_remaining_minutes': '31.65'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.988', 'eval_runtime': '15.76', 'eval_samples_per_second': '604.5', 'eval_steps_per_second': '4.758', 'epoch': '0.1348', 'train/total_time_seconds': '376.8', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '553.9', 'train/estimated_remaining_minutes': '31.4'} +{'loss': '15.99', 'grad_norm': '4.188', 'learning_rate': '0.001', 'epoch': '0.1402', 'train/total_time_seconds': '391.8', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '572.7', 'train/estimated_remaining_minutes': '31.14'} +{'loss': '15.49', 'grad_norm': '2.156', 'learning_rate': '0.001', 'epoch': '0.151', 'train/total_time_seconds': '421.7', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '610.2', 'train/estimated_remaining_minutes': '30.62'} +{'loss': '14.99', 'grad_norm': '1.516', 'learning_rate': '0.001', 'epoch': '0.1618', 'train/total_time_seconds': '451.5', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '647.6', 'train/estimated_remaining_minutes': '30.1'} +{'eval_loss': '3.689', 'eval_runtime': '15.74', 'eval_samples_per_second': '605.3', 'eval_steps_per_second': '4.765', 'epoch': '0.1618', 'train/total_time_seconds': '451.5', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '663.3', 'train/estimated_remaining_minutes': '30.1'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 14.87it/s] + 27%|██████████▍ | 400/1500 [14:43<34:11, 1.86s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '14.5', 'grad_norm': '1.875', 'learning_rate': '0.001', 'epoch': '0.1726', 'train/total_time_seconds': '481.3', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '701.1', 'train/estimated_remaining_minutes': '29.58'} +{'loss': '14.09', 'grad_norm': '2.656', 'learning_rate': '0.001', 'epoch': '0.1833', 'train/total_time_seconds': '511.2', 'train/time_per_step_avg': '1.494', 'train/epoch_time_elapsed': '738.7', 'train/estimated_remaining_minutes': '29.07'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.426', 'eval_runtime': '15.72', 'eval_samples_per_second': '606', 'eval_steps_per_second': '4.77', 'epoch': '0.1887', 'train/total_time_seconds': '526.1', 'train/time_per_step_avg': '1.492', 'train/epoch_time_elapsed': '773.1', 'train/estimated_remaining_minutes': '28.81'} +{'loss': '13.7', 'grad_norm': '2.859', 'learning_rate': '0.001', 'epoch': '0.1941', 'train/total_time_seconds': '541', 'train/time_per_step_avg': '1.492', 'train/epoch_time_elapsed': '791.8', 'train/estimated_remaining_minutes': '28.55'} +{'loss': '13.35', 'grad_norm': '1.562', 'learning_rate': '0.001', 'epoch': '0.2049', 'train/total_time_seconds': '571.1', 'train/time_per_step_avg': '1.494', 'train/epoch_time_elapsed': '829.6', 'train/estimated_remaining_minutes': '28.05'} +{'loss': '12.98', 'grad_norm': '1.531', 'learning_rate': '0.001', 'epoch': '0.2157', 'train/total_time_seconds': '601', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '866.9', 'train/estimated_remaining_minutes': '27.54'} +{'eval_loss': '3.222', 'eval_runtime': '15.78', 'eval_samples_per_second': '603.6', 'eval_steps_per_second': '4.752', 'epoch': '0.2157', 'train/total_time_seconds': '601', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '882.7', 'train/estimated_remaining_minutes': '27.54'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 17.38it/s] + 33%|█████████████ | 500/1500 [18:23<31:25, 1.89s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '12.66', 'grad_norm': '2.312', 'learning_rate': '0.001', 'epoch': '0.2265', 'train/total_time_seconds': '630.8', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '920.4', 'train/estimated_remaining_minutes': '27.03'} +{'loss': '12.41', 'grad_norm': '1.805', 'learning_rate': '0.001', 'epoch': '0.2373', 'train/total_time_seconds': '660.7', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '958.1', 'train/estimated_remaining_minutes': '26.53'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.017', 'eval_runtime': '15.63', 'eval_samples_per_second': '609.6', 'eval_steps_per_second': '4.799', 'epoch': '0.2427', 'train/total_time_seconds': '675.7', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '992.8', 'train/estimated_remaining_minutes': '26.28'} +{'loss': '12.05', 'grad_norm': '1.867', 'learning_rate': '0.001', 'epoch': '0.248', 'train/total_time_seconds': '690.6', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '1011', 'train/estimated_remaining_minutes': '26.02'} +{'loss': '11.73', 'grad_norm': '1.406', 'learning_rate': '0.001', 'epoch': '0.2588', 'train/total_time_seconds': '720.5', 'train/time_per_step_avg': '1.494', 'train/epoch_time_elapsed': '1049', 'train/estimated_remaining_minutes': '25.52'} +{'loss': '11.41', 'grad_norm': '1.531', 'learning_rate': '0.001', 'epoch': '0.2696', 'train/total_time_seconds': '750.4', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '1087', 'train/estimated_remaining_minutes': '25.01'} +{'eval_loss': '2.821', 'eval_runtime': '16', 'eval_samples_per_second': '595.4', 'eval_steps_per_second': '4.687', 'epoch': '0.2696', 'train/total_time_seconds': '750.4', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '1103', 'train/estimated_remaining_minutes': '25.01'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 3.41it/s] +/mnt/data/zainulabideen/zain-exp/notebooks/Activation/exp.py:487: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1831.) + "std": tensor.std().item(), + 40%|███████████████▌ | 600/1500 [22:06<28:08, 1.88s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '11.16', 'grad_norm': '1.398', 'learning_rate': '0.001', 'epoch': '0.2804', 'train/total_time_seconds': '782.9', 'train/time_per_step_avg': '1.521', 'train/epoch_time_elapsed': '1144', 'train/estimated_remaining_minutes': '24.59', 'train/global/act/norm': '2.108e+05', 'train/global/act/mean': '-0.04442', 'train/global/act/std': '0.9892', 'train/global/act/max_abs': '30.5', 'train/global/act/frac_near_dtype_limit': '0', 'train/global/act/frac_near_user_limit': '0', 'train/global/grad/norm': '0.8868', 'train/global/grad/mean': '-3.285e-08', 'train/global/grad/std': '0.000111', 'train/global/grad/max_abs': '0.02441', 'train/global/grad/frac_near_dtype_limit': '0', 'train/global/grad/frac_near_user_limit': '0', 'train/global/param/norm': '204.6', 'train/global/param/mean': '0.001524', 'train/global/param/std': '0.05121', 'train/global/param/max_abs': '1', 'train/global/param/frac_near_dtype_limit': '0', 'train/global/param/frac_near_user_limit': '0', 'train/layer__model_layers_49/param/norm': '19.65', 'train/layer__model_layers_49/param/mean': '0.001461', 'train/layer__model_layers_49/param/std': '0.0485', 'train/layer__model_layers_49/param/max_abs': '1', 'train/layer__model_layers_49/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_49/param/frac_near_user_limit': '0', 'train/layer__model_layers_7/param/norm': '18.96', 'train/layer__model_layers_7/param/mean': '0.001547', 'train/layer__model_layers_7/param/std': '0.04675', 'train/layer__model_layers_7/param/max_abs': '1', 'train/layer__model_layers_7/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_7/param/frac_near_user_limit': '0', 'train/layer__model_layers_86/param/norm': '22.84', 'train/layer__model_layers_86/param/mean': '0.001594', 'train/layer__model_layers_86/param/std': '0.05639', 'train/layer__model_layers_86/param/max_abs': '1', 'train/layer__model_layers_86/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_86/param/frac_near_user_limit': '0', 'train/layer_model_layers_4/act/norm': '2.117e+04', 'train/layer_model_layers_4/act/mean': '-0.001465', 'train/layer_model_layers_4/act/std': '0.9766', 'train/layer_model_layers_4/act/max_abs': '17.88', 'train/layer_model_layers_4/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_4/act/frac_near_user_limit': '0', 'train/layer_model_layers_4/grad/norm': '0.06447', 'train/layer_model_layers_4/grad/mean': '8.797e-08', 'train/layer_model_layers_4/grad/std': '7.95e-05', 'train/layer_model_layers_4/grad/max_abs': '0.001686', 'train/layer_model_layers_4/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_4/grad/frac_near_user_limit': '0', 'train/layer_model_layers_40/act/norm': '1.85e+04', 'train/layer_model_layers_40/act/mean': '0.001928', 'train/layer_model_layers_40/act/std': '0.8529', 'train/layer_model_layers_40/act/max_abs': '20.25', 'train/layer_model_layers_40/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_40/act/frac_near_user_limit': '0', 'train/layer_model_layers_40/grad/norm': '0.0276', 'train/layer_model_layers_40/grad/mean': '3.439e-08', 'train/layer_model_layers_40/grad/std': '3.406e-05', 'train/layer_model_layers_40/grad/max_abs': '0.0007095', 'train/layer_model_layers_40/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_40/grad/frac_near_user_limit': '0', 'train/layer_model_layers_34/act/norm': '1.832e+04', 'train/layer_model_layers_34/act/mean': '0.002268', 'train/layer_model_layers_34/act/std': '0.8442', 'train/layer_model_layers_34/act/max_abs': '20.25', 'train/layer_model_layers_34/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_34/act/frac_near_user_limit': '0', 'train/layer_model_layers_34/grad/norm': '0.02417', 'train/layer_model_layers_34/grad/mean': '1.063e-07', 'train/layer_model_layers_34/grad/std': '2.985e-05', 'train/layer_model_layers_34/grad/max_abs': '0.0006027', 'train/layer_model_layers_34/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_34/grad/frac_near_user_limit': '0', 'train/layer_model_layers_27/act/norm': '1.909e+04', 'train/layer_model_layers_27/act/mean': '-0.001571', 'train/layer +{'loss': '10.94', 'grad_norm': '1.609', 'learning_rate': '0.001', 'epoch': '0.2912', 'train/total_time_seconds': '812.8', 'train/time_per_step_avg': '1.521', 'train/epoch_time_elapsed': '1181', 'train/estimated_remaining_minutes': '24.08'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.668', 'eval_runtime': '15.7', 'eval_samples_per_second': '606.9', 'eval_steps_per_second': '4.777', 'epoch': '0.2966', 'train/total_time_seconds': '827.8', 'train/time_per_step_avg': '1.521', 'train/epoch_time_elapsed': '1216', 'train/estimated_remaining_minutes': '23.83'} +{'loss': '10.69', 'grad_norm': '1.742', 'learning_rate': '0.001', 'epoch': '0.302', 'train/total_time_seconds': '842.7', 'train/time_per_step_avg': '1.521', 'train/epoch_time_elapsed': '1235', 'train/estimated_remaining_minutes': '23.58'} +{'loss': '10.45', 'grad_norm': '1.227', 'learning_rate': '0.001', 'epoch': '0.3128', 'train/total_time_seconds': '872.5', 'train/time_per_step_avg': '1.52', 'train/epoch_time_elapsed': '1273', 'train/estimated_remaining_minutes': '23.07'} +{'loss': '10.23', 'grad_norm': '1.359', 'learning_rate': '0.001', 'epoch': '0.3235', 'train/total_time_seconds': '902.4', 'train/time_per_step_avg': '1.519', 'train/epoch_time_elapsed': '1310', 'train/estimated_remaining_minutes': '22.56'} +{'eval_loss': '2.535', 'eval_runtime': '15.93', 'eval_samples_per_second': '597.9', 'eval_steps_per_second': '4.707', 'epoch': '0.3235', 'train/total_time_seconds': '902.4', 'train/time_per_step_avg': '1.519', 'train/epoch_time_elapsed': '1326', 'train/estimated_remaining_minutes': '22.56'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 16.86it/s] + 47%|██████████████████▏ | 700/1500 [25:47<24:58, 1.87s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '10.04', 'grad_norm': '1.656', 'learning_rate': '0.001', 'epoch': '0.3343', 'train/total_time_seconds': '932.5', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '1364', 'train/estimated_remaining_minutes': '22.06'} +{'loss': '9.852', 'grad_norm': '1.203', 'learning_rate': '0.001', 'epoch': '0.3451', 'train/total_time_seconds': '962.4', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '1402', 'train/estimated_remaining_minutes': '21.55'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.426', 'eval_runtime': '15.89', 'eval_samples_per_second': '599.5', 'eval_steps_per_second': '4.719', 'epoch': '0.3505', 'train/total_time_seconds': '977.4', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '1437', 'train/estimated_remaining_minutes': '21.3'} +{'loss': '9.675', 'grad_norm': '1.219', 'learning_rate': '0.001', 'epoch': '0.3559', 'train/total_time_seconds': '992.3', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '1455', 'train/estimated_remaining_minutes': '21.05'} +{'loss': '9.502', 'grad_norm': '1.18', 'learning_rate': '0.001', 'epoch': '0.3667', 'train/total_time_seconds': '1022', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '1493', 'train/estimated_remaining_minutes': '20.54'} +{'loss': '9.35', 'grad_norm': '1.031', 'learning_rate': '0.001', 'epoch': '0.3775', 'train/total_time_seconds': '1052', 'train/time_per_step_avg': '1.497', 'train/epoch_time_elapsed': '1531', 'train/estimated_remaining_minutes': '20.04'} +{'eval_loss': '2.323', 'eval_runtime': '15.89', 'eval_samples_per_second': '599.7', 'eval_steps_per_second': '4.721', 'epoch': '0.3775', 'train/total_time_seconds': '1052', 'train/time_per_step_avg': '1.497', 'train/epoch_time_elapsed': '1546', 'train/estimated_remaining_minutes': '20.04'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 15.13it/s] + 53%|████████████████████▊ | 800/1500 [29:27<22:14, 1.91s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '9.192', 'grad_norm': '1.344', 'learning_rate': '0.001', 'epoch': '0.3882', 'train/total_time_seconds': '1082', 'train/time_per_step_avg': '1.494', 'train/epoch_time_elapsed': '1584', 'train/estimated_remaining_minutes': '19.53'} +{'loss': '9.084', 'grad_norm': '1.227', 'learning_rate': '0.001', 'epoch': '0.399', 'train/total_time_seconds': '1112', 'train/time_per_step_avg': '1.493', 'train/epoch_time_elapsed': '1622', 'train/estimated_remaining_minutes': '19.03'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.241', 'eval_runtime': '16', 'eval_samples_per_second': '595.5', 'eval_steps_per_second': '4.688', 'epoch': '0.4044', 'train/total_time_seconds': '1127', 'train/time_per_step_avg': '1.493', 'train/epoch_time_elapsed': '1657', 'train/estimated_remaining_minutes': '18.78'} +{'loss': '8.964', 'grad_norm': '1.156', 'learning_rate': '0.001', 'epoch': '0.4098', 'train/total_time_seconds': '1142', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '1676', 'train/estimated_remaining_minutes': '18.53'} +{'loss': '8.877', 'grad_norm': '1.297', 'learning_rate': '0.001', 'epoch': '0.4206', 'train/total_time_seconds': '1172', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '1714', 'train/estimated_remaining_minutes': '18.03'} +{'loss': '8.725', 'grad_norm': '1.203', 'learning_rate': '0.001', 'epoch': '0.4314', 'train/total_time_seconds': '1202', 'train/time_per_step_avg': '1.497', 'train/epoch_time_elapsed': '1751', 'train/estimated_remaining_minutes': '17.53'} +{'eval_loss': '2.179', 'eval_runtime': '15.97', 'eval_samples_per_second': '596.5', 'eval_steps_per_second': '4.696', 'epoch': '0.4314', 'train/total_time_seconds': '1202', 'train/time_per_step_avg': '1.497', 'train/epoch_time_elapsed': '1767', 'train/estimated_remaining_minutes': '17.53'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 17.56it/s] + 60%|███████████████████████▍ | 900/1500 [33:08<19:04, 1.91s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '8.653', 'grad_norm': '1.078', 'learning_rate': '0.001', 'epoch': '0.4422', 'train/total_time_seconds': '1232', 'train/time_per_step_avg': '1.498', 'train/epoch_time_elapsed': '1805', 'train/estimated_remaining_minutes': '17.02'} +{'loss': '8.572', 'grad_norm': '1.219', 'learning_rate': '0.001', 'epoch': '0.453', 'train/total_time_seconds': '1262', 'train/time_per_step_avg': '1.499', 'train/epoch_time_elapsed': '1843', 'train/estimated_remaining_minutes': '16.52'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.128', 'eval_runtime': '15.83', 'eval_samples_per_second': '601.9', 'eval_steps_per_second': '4.738', 'epoch': '0.4583', 'train/total_time_seconds': '1277', 'train/time_per_step_avg': '1.499', 'train/epoch_time_elapsed': '1878', 'train/estimated_remaining_minutes': '16.27'} +{'loss': '8.466', 'grad_norm': '1.094', 'learning_rate': '0.001', 'epoch': '0.4637', 'train/total_time_seconds': '1291', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '1896', 'train/estimated_remaining_minutes': '16.02'} +{'loss': '8.38', 'grad_norm': '1.008', 'learning_rate': '0.001', 'epoch': '0.4745', 'train/total_time_seconds': '1321', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '1934', 'train/estimated_remaining_minutes': '15.52'} +{'loss': '8.304', 'grad_norm': '0.9336', 'learning_rate': '0.001', 'epoch': '0.4853', 'train/total_time_seconds': '1352', 'train/time_per_step_avg': '1.498', 'train/epoch_time_elapsed': '1972', 'train/estimated_remaining_minutes': '15.02'} +{'eval_loss': '2.072', 'eval_runtime': '15.82', 'eval_samples_per_second': '602.1', 'eval_steps_per_second': '4.74', 'epoch': '0.4853', 'train/total_time_seconds': '1352', 'train/time_per_step_avg': '1.498', 'train/epoch_time_elapsed': '1988', 'train/estimated_remaining_minutes': '15.02'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 13.94it/s] + 67%|█████████████████████████▎ | 1000/1500 [36:49<15:56, 1.91s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '8.227', 'grad_norm': '1.008', 'learning_rate': '0.001', 'epoch': '0.4961', 'train/total_time_seconds': '1381', 'train/time_per_step_avg': '1.497', 'train/epoch_time_elapsed': '2026', 'train/estimated_remaining_minutes': '14.51'} +{'loss': '8.188', 'grad_norm': '0.8984', 'learning_rate': '0.001', 'epoch': '0.5069', 'train/total_time_seconds': '1411', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '2064', 'train/estimated_remaining_minutes': '14.01'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.034', 'eval_runtime': '15.98', 'eval_samples_per_second': '596.1', 'eval_steps_per_second': '4.693', 'epoch': '0.5123', 'train/total_time_seconds': '1426', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '2098', 'train/estimated_remaining_minutes': '13.76'} +{'loss': '8.129', 'grad_norm': '0.9219', 'learning_rate': '0.001', 'epoch': '0.5177', 'train/total_time_seconds': '1441', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '2117', 'train/estimated_remaining_minutes': '13.51'} +{'loss': '8.076', 'grad_norm': '0.9453', 'learning_rate': '0.001', 'epoch': '0.5284', 'train/total_time_seconds': '1471', 'train/time_per_step_avg': '1.497', 'train/epoch_time_elapsed': '2155', 'train/estimated_remaining_minutes': '13.01'} +{'loss': '8.016', 'grad_norm': '1.039', 'learning_rate': '0.001', 'epoch': '0.5392', 'train/total_time_seconds': '1501', 'train/time_per_step_avg': '1.494', 'train/epoch_time_elapsed': '2193', 'train/estimated_remaining_minutes': '12.51'} +{'eval_loss': '2.001', 'eval_runtime': '15.8', 'eval_samples_per_second': '603.1', 'eval_steps_per_second': '4.748', 'epoch': '0.5392', 'train/total_time_seconds': '1501', 'train/time_per_step_avg': '1.494', 'train/epoch_time_elapsed': '2209', 'train/estimated_remaining_minutes': '12.51'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 18.81it/s] +/mnt/data/zainulabideen/zain-exp/notebooks/Activation/exp.py:487: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1831.) + "std": tensor.std().item(), + 73%|███████████████████████████▊ | 1100/1500 [40:32<12:34, 1.89s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '7.937', 'grad_norm': '0.9922', 'learning_rate': '0.001', 'epoch': '0.55', 'train/total_time_seconds': '1533', 'train/time_per_step_avg': '1.519', 'train/epoch_time_elapsed': '2249', 'train/estimated_remaining_minutes': '12.03', 'train/global/act/norm': '1.788e+05', 'train/global/act/mean': '-0.05405', 'train/global/act/std': '0.8384', 'train/global/act/max_abs': '26.12', 'train/global/act/frac_near_dtype_limit': '0', 'train/global/act/frac_near_user_limit': '0', 'train/global/grad/norm': '0.6355', 'train/global/grad/mean': '-1.679e-08', 'train/global/grad/std': '7.957e-05', 'train/global/grad/max_abs': '0.01282', 'train/global/grad/frac_near_dtype_limit': '0', 'train/global/grad/frac_near_user_limit': '0', 'train/global/param/norm': '231.5', 'train/global/param/mean': '0.001526', 'train/global/param/std': '0.05795', 'train/global/param/max_abs': '1', 'train/global/param/frac_near_dtype_limit': '0', 'train/global/param/frac_near_user_limit': '0', 'train/layer__model_layers_49/param/norm': '21.78', 'train/layer__model_layers_49/param/mean': '0.001486', 'train/layer__model_layers_49/param/std': '0.05377', 'train/layer__model_layers_49/param/max_abs': '1', 'train/layer__model_layers_49/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_49/param/frac_near_user_limit': '0', 'train/layer__model_layers_7/param/norm': '19.6', 'train/layer__model_layers_7/param/mean': '0.001541', 'train/layer__model_layers_7/param/std': '0.04837', 'train/layer__model_layers_7/param/max_abs': '1', 'train/layer__model_layers_7/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_7/param/frac_near_user_limit': '0', 'train/layer__model_layers_86/param/norm': '25.44', 'train/layer__model_layers_86/param/mean': '0.001714', 'train/layer__model_layers_86/param/std': '0.06281', 'train/layer__model_layers_86/param/max_abs': '1', 'train/layer__model_layers_86/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_86/param/frac_near_user_limit': '0', 'train/layer_model_layers_4/act/norm': '1.538e+04', 'train/layer_model_layers_4/act/mean': '-0.002358', 'train/layer_model_layers_4/act/std': '0.7093', 'train/layer_model_layers_4/act/max_abs': '8.812', 'train/layer_model_layers_4/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_4/act/frac_near_user_limit': '0', 'train/layer_model_layers_4/grad/norm': '0.06301', 'train/layer_model_layers_4/grad/mean': '9.569e-08', 'train/layer_model_layers_4/grad/std': '7.779e-05', 'train/layer_model_layers_4/grad/max_abs': '0.001793', 'train/layer_model_layers_4/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_4/grad/frac_near_user_limit': '0', 'train/layer_model_layers_40/act/norm': '1.37e+04', 'train/layer_model_layers_40/act/mean': '-0.002146', 'train/layer_model_layers_40/act/std': '0.6324', 'train/layer_model_layers_40/act/max_abs': '9.375', 'train/layer_model_layers_40/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_40/act/frac_near_user_limit': '0', 'train/layer_model_layers_40/grad/norm': '0.03881', 'train/layer_model_layers_40/grad/mean': '-1.765e-08', 'train/layer_model_layers_40/grad/std': '4.79e-05', 'train/layer_model_layers_40/grad/max_abs': '0.001312', 'train/layer_model_layers_40/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_40/grad/frac_near_user_limit': '0', 'train/layer_model_layers_34/act/norm': '1.354e+04', 'train/layer_model_layers_34/act/mean': '0.002037', 'train/layer_model_layers_34/act/std': '0.6245', 'train/layer_model_layers_34/act/max_abs': '9.312', 'train/layer_model_layers_34/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_34/act/frac_near_user_limit': '0', 'train/layer_model_layers_34/grad/norm': '0.03821', 'train/layer_model_layers_34/grad/mean': '1.768e-08', 'train/layer_model_layers_34/grad/std': '4.715e-05', 'train/layer_model_layers_34/grad/max_abs': '0.001289', 'train/layer_model_layers_34/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_34/grad/frac_near_user_limit': '0', 'train/layer_model_layers_27/act/norm': '1.372e+04', 'train/layer_model_layers_27/act/mean': '-0.004381', 'train/laye +{'loss': '7.884', 'grad_norm': '0.9062', 'learning_rate': '0.001', 'epoch': '0.5608', 'train/total_time_seconds': '1563', 'train/time_per_step_avg': '1.519', 'train/epoch_time_elapsed': '2287', 'train/estimated_remaining_minutes': '11.52'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '1.969', 'eval_runtime': '15.81', 'eval_samples_per_second': '602.8', 'eval_steps_per_second': '4.745', 'epoch': '0.5662', 'train/total_time_seconds': '1578', 'train/time_per_step_avg': '1.519', 'train/epoch_time_elapsed': '2322', 'train/estimated_remaining_minutes': '11.27'} +{'loss': '7.831', 'grad_norm': '0.9609', 'learning_rate': '0.001', 'epoch': '0.5716', 'train/total_time_seconds': '1593', 'train/time_per_step_avg': '1.519', 'train/epoch_time_elapsed': '2340', 'train/estimated_remaining_minutes': '11.02'} +{'loss': '7.804', 'grad_norm': '0.9102', 'learning_rate': '0.001', 'epoch': '0.5824', 'train/total_time_seconds': '1623', 'train/time_per_step_avg': '1.519', 'train/epoch_time_elapsed': '2378', 'train/estimated_remaining_minutes': '10.52'} +{'loss': '7.771', 'grad_norm': '0.8984', 'learning_rate': '0.001', 'epoch': '0.5932', 'train/total_time_seconds': '1653', 'train/time_per_step_avg': '1.519', 'train/epoch_time_elapsed': '2416', 'train/estimated_remaining_minutes': '10.02'} +{'eval_loss': '1.945', 'eval_runtime': '15.89', 'eval_samples_per_second': '599.7', 'eval_steps_per_second': '4.721', 'epoch': '0.5932', 'train/total_time_seconds': '1653', 'train/time_per_step_avg': '1.519', 'train/epoch_time_elapsed': '2432', 'train/estimated_remaining_minutes': '10.02'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 13.85it/s] + 80%|██████████████████████████████▍ | 1200/1500 [44:13<09:25, 1.89s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '7.725', 'grad_norm': '1.133', 'learning_rate': '0.001', 'epoch': '0.6039', 'train/total_time_seconds': '1683', 'train/time_per_step_avg': '1.494', 'train/epoch_time_elapsed': '2470', 'train/estimated_remaining_minutes': '9.516'} +{'loss': '7.69', 'grad_norm': '0.8555', 'learning_rate': '0.001', 'epoch': '0.6147', 'train/total_time_seconds': '1713', 'train/time_per_step_avg': '1.497', 'train/epoch_time_elapsed': '2508', 'train/estimated_remaining_minutes': '9.015'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '1.92', 'eval_runtime': '16.04', 'eval_samples_per_second': '593.8', 'eval_steps_per_second': '4.675', 'epoch': '0.6201', 'train/total_time_seconds': '1728', 'train/time_per_step_avg': '1.497', 'train/epoch_time_elapsed': '2543', 'train/estimated_remaining_minutes': '8.764'} +{'loss': '7.65', 'grad_norm': '0.8359', 'learning_rate': '0.001', 'epoch': '0.6255', 'train/total_time_seconds': '1743', 'train/time_per_step_avg': '1.497', 'train/epoch_time_elapsed': '2562', 'train/estimated_remaining_minutes': '8.513'} +{'loss': '7.613', 'grad_norm': '0.9141', 'learning_rate': '0.001', 'epoch': '0.6363', 'train/total_time_seconds': '1773', 'train/time_per_step_avg': '1.496', 'train/epoch_time_elapsed': '2599', 'train/estimated_remaining_minutes': '8.011'} +{'loss': '7.558', 'grad_norm': '0.8867', 'learning_rate': '0.001', 'epoch': '0.6471', 'train/total_time_seconds': '1802', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '2637', 'train/estimated_remaining_minutes': '7.51'} +{'eval_loss': '1.897', 'eval_runtime': '15.85', 'eval_samples_per_second': '601.2', 'eval_steps_per_second': '4.733', 'epoch': '0.6471', 'train/total_time_seconds': '1802', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '2653', 'train/estimated_remaining_minutes': '7.51'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 16.66it/s] + 87%|████████████████████████████████▉ | 1300/1500 [47:53<06:17, 1.89s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '7.555', 'grad_norm': '0.9883', 'learning_rate': '0.001', 'epoch': '0.6579', 'train/total_time_seconds': '1832', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '2691', 'train/estimated_remaining_minutes': '7.009'} +{'loss': '7.521', 'grad_norm': '0.8711', 'learning_rate': '0.001', 'epoch': '0.6686', 'train/total_time_seconds': '1862', 'train/time_per_step_avg': '1.493', 'train/epoch_time_elapsed': '2728', 'train/estimated_remaining_minutes': '6.507'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '1.878', 'eval_runtime': '15.95', 'eval_samples_per_second': '597.2', 'eval_steps_per_second': '4.701', 'epoch': '0.674', 'train/total_time_seconds': '1877', 'train/time_per_step_avg': '1.493', 'train/epoch_time_elapsed': '2763', 'train/estimated_remaining_minutes': '6.257'} +{'loss': '7.483', 'grad_norm': '0.9219', 'learning_rate': '0.001', 'epoch': '0.6794', 'train/total_time_seconds': '1892', 'train/time_per_step_avg': '1.492', 'train/epoch_time_elapsed': '2782', 'train/estimated_remaining_minutes': '6.006'} +{'loss': '7.478', 'grad_norm': '0.9141', 'learning_rate': '0.001', 'epoch': '0.6902', 'train/total_time_seconds': '1922', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '2819', 'train/estimated_remaining_minutes': '5.506'} +{'loss': '7.426', 'grad_norm': '0.8438', 'learning_rate': '0.001', 'epoch': '0.701', 'train/total_time_seconds': '1952', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '2857', 'train/estimated_remaining_minutes': '5.005'} +{'eval_loss': '1.861', 'eval_runtime': '15.84', 'eval_samples_per_second': '601.6', 'eval_steps_per_second': '4.736', 'epoch': '0.701', 'train/total_time_seconds': '1952', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '2873', 'train/estimated_remaining_minutes': '5.005'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 13.14it/s] + 93%|███████████████████████████████████▍ | 1400/1500 [51:42<03:07, 1.87s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '7.397', 'grad_norm': '0.875', 'learning_rate': '0.001', 'epoch': '0.7118', 'train/total_time_seconds': '1982', 'train/time_per_step_avg': '1.495', 'train/epoch_time_elapsed': '2910', 'train/estimated_remaining_minutes': '4.504'} +{'loss': '7.381', 'grad_norm': '0.8477', 'learning_rate': '0.001', 'epoch': '0.7226', 'train/total_time_seconds': '2013', 'train/time_per_step_avg': '1.504', 'train/epoch_time_elapsed': '2949', 'train/estimated_remaining_minutes': '4.005'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '1.846', 'eval_runtime': '16.77', 'eval_samples_per_second': '568', 'eval_steps_per_second': '4.471', 'epoch': '0.728', 'train/total_time_seconds': '2032', 'train/time_per_step_avg': '1.553', 'train/epoch_time_elapsed': '2989', 'train/estimated_remaining_minutes': '3.763'} +{'loss': '7.354', 'grad_norm': '0.9492', 'learning_rate': '0.001', 'epoch': '0.7334', 'train/total_time_seconds': '2049', 'train/time_per_step_avg': '1.574', 'train/epoch_time_elapsed': '3010', 'train/estimated_remaining_minutes': '3.516'} +{'loss': '7.347', 'grad_norm': '0.7656', 'learning_rate': '0.001', 'epoch': '0.7441', 'train/total_time_seconds': '2079', 'train/time_per_step_avg': '1.572', 'train/epoch_time_elapsed': '3048', 'train/estimated_remaining_minutes': '3.013'} +{'loss': '7.306', 'grad_norm': '0.8125', 'learning_rate': '0.001', 'epoch': '0.7549', 'train/total_time_seconds': '2109', 'train/time_per_step_avg': '1.574', 'train/epoch_time_elapsed': '3086', 'train/estimated_remaining_minutes': '2.511'} +{'eval_loss': '1.832', 'eval_runtime': '16.11', 'eval_samples_per_second': '591.5', 'eval_steps_per_second': '4.657', 'epoch': '0.7549', 'train/total_time_seconds': '2109', 'train/time_per_step_avg': '1.574', 'train/epoch_time_elapsed': '3102', 'train/estimated_remaining_minutes': '2.511'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 18.65it/s] +100%|██████████████████████████████████████| 1500/1500 [56:00<00:00, 2.45s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '7.275', 'grad_norm': '0.8125', 'learning_rate': '0.001', 'epoch': '0.7657', 'train/total_time_seconds': '2139', 'train/time_per_step_avg': '1.575', 'train/epoch_time_elapsed': '3140', 'train/estimated_remaining_minutes': '2.009'} +{'loss': '7.275', 'grad_norm': '0.75', 'learning_rate': '0.001', 'epoch': '0.7765', 'train/total_time_seconds': '2169', 'train/time_per_step_avg': '1.568', 'train/epoch_time_elapsed': '3178', 'train/estimated_remaining_minutes': '1.507'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '1.819', 'eval_runtime': '19.27', 'eval_samples_per_second': '494.3', 'eval_steps_per_second': '3.891', 'epoch': '0.7819', 'train/total_time_seconds': '2189', 'train/time_per_step_avg': '1.569', 'train/epoch_time_elapsed': '3221', 'train/estimated_remaining_minutes': '1.258'} +{'loss': '7.239', 'grad_norm': '0.832', 'learning_rate': '0.001', 'epoch': '0.7873', 'train/total_time_seconds': '2210', 'train/time_per_step_avg': '1.604', 'train/epoch_time_elapsed': '3245', 'train/estimated_remaining_minutes': '1.009'} +{'loss': '7.227', 'grad_norm': '0.8672', 'learning_rate': '0.001', 'epoch': '0.7981', 'train/total_time_seconds': '2249', 'train/time_per_step_avg': '1.698', 'train/epoch_time_elapsed': '3293', 'train/estimated_remaining_minutes': '0.5065'} +{'loss': '7.199', 'grad_norm': '0.8008', 'learning_rate': '0.001', 'epoch': '0.8088', 'train/total_time_seconds': '2290', 'train/time_per_step_avg': '1.804', 'train/epoch_time_elapsed': '3341', 'train/estimated_remaining_minutes': '0'} +{'eval_loss': '1.807', 'eval_runtime': '18.87', 'eval_samples_per_second': '504.9', 'eval_steps_per_second': '3.975', 'epoch': '0.8088', 'train/total_time_seconds': '2290', 'train/time_per_step_avg': '1.804', 'train/epoch_time_elapsed': '3360', 'train/estimated_remaining_minutes': '0'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 11.91it/s] +100%|██████████████████████████████████████| 1500/1500 [56:01<00:00, 2.24s/it] +{'train_runtime': '3361', 'train_samples_per_second': '228.5', 'train_steps_per_second': '0.446', 'train_loss': '11.32', 'epoch': '0.8088', 'train/total_time_seconds': '2290', 'train/time_per_step_avg': '1.804', 'train/epoch_time_elapsed': '3360', 'train/estimated_remaining_minutes': '0'} +100%|██████████████████████████████████████████| 75/75 [00:18<00:00, 4.06it/s] +[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 15.65it/s] +[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 17.96it/s] +Found 7 files to upload + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████░░░░░░░░░░░░ 3 / 7 + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████░░░░░░░░░░░░ 3 / 7 + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████████████████ 7 / 7 ✓ +[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 16.32it/s] +Found 7 files to upload + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████████████████ 7 / 7 ✓ +No files have been modified since last commit. Skipping to prevent empty commit. diff --git a/wandb/run-20260809_050050-59pftr14/files/requirements.txt b/wandb/run-20260809_050050-59pftr14/files/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..123b15ebf857859624f7f4332e92341c8ef13fdf --- /dev/null +++ b/wandb/run-20260809_050050-59pftr14/files/requirements.txt @@ -0,0 +1,149 @@ +asttokens==3.0.1 +comm==0.2.3 +debugpy==1.8.21 +decorator==5.3.1 +executing==2.2.1 +nest-asyncio==1.6.0 +parso==0.8.7 +platformdirs==4.11.0 +psutil==7.2.2 +ptyprocess==0.7.0 +pure_eval==0.2.3 +Pygments==2.20.0 +pyzmq==27.1.0 +setuptools==83.0.0 +six==1.17.0 +tornado==6.5.7 +traitlets==5.15.0 +fsspec==2026.4.0 +wcwidth==0.8.2 +ipython_pygments_lexers==1.1.1 +jedi==0.20.0 +jupyter_core==5.9.1 +matplotlib-inline==0.2.2 +pexpect==4.9.0 +prompt_toolkit==3.0.53 +python-dateutil==2.9.0.post0 +stack_data==0.6.3 +wheel==0.47.0 +jupyter_client==8.9.1 +pip==26.1.2 +ipython==9.15.0 +ipykernel==7.2.0 +threadpoolctl==3.6.0 +pyparsing==3.3.2 +typing_extensions==4.15.0 +Jinja2==3.1.6 +narwhals==2.24.0 +kiwisolver==1.5.0 +joblib==1.5.3 +fonttools==4.63.0 +cycler==0.12.1 +scipy==1.17.1 +pandas==3.0.5 +contourpy==1.3.3 +scikit-learn==1.9.0 +matplotlib==3.11.1 +urllib3==2.7.0 +tqdm==4.70.0 +idna==3.18 +charset-normalizer==3.4.9 +certifi==2026.7.22 +requests==2.34.2 +seaborn==0.13.2 +uv==0.12.0 +shellingham==1.5.4 +mpmath==1.3.0 +attrs==26.1.0 +hf-xet==1.5.2 +nvidia-nccl-cu12==2.21.5 +MarkupSafe==3.0.3 +regex==2026.7.19 +importlib_metadata==9.0.0 +httpcore==1.0.9 +annotated-doc==0.0.5 +multidict==6.7.1 +aiohttp==3.14.3 +aiosignal==1.4.0 +xxhash==3.8.1 +aiohappyeyeballs==2.7.1 +mdurl==0.1.2 +cuda-toolkit==13.0.3.0 +networkx==3.6.1 +PyYAML==6.0.3 +nvidia-cufile==1.15.1.6 +typer==0.27.0 +torchaudio==2.6.0+cu124 +rich==15.0.0 +nvidia-cufft-cu12==11.2.1.3 +h11==0.16.0 +dill==0.4.1 +cuda-pathfinder==1.6.0 +filelock==3.29.0 +nvidia-nvtx-cu12==12.4.127 +httpx==0.28.1 +anyio==4.14.2 +numpy==2.4.4 +yarl==1.24.5 +click==8.4.2 +triton==3.2.0 +frozenlist==1.8.0 +zipp==4.1.0 +propcache==0.5.2 +tokenizers==0.22.2 +markdown-it-py==4.2.0 +nvidia-cuda-runtime==13.0.96 +cuda-bindings==13.3.1 +nvidia-cuda-cupti==13.0.85 +torch==2.6.0+cu124 +multiprocess==0.70.19 +pillow==12.2.0 +transformers==5.15.0.dev0 +wandb==0.28.1 +nvidia-curand==10.4.0.35 +sympy==1.13.1 +nvidia-cusparse==12.6.3.3 +nvidia-cuda-nvrtc==13.0.88 +typing-inspection==0.4.2 +nvidia-cusolver==12.0.4.66 +nvidia-cufft==12.0.0.61 +nvidia-cudnn-cu13==9.20.0.48 +nvidia-cublas==13.1.1.3 +pyarrow==25.0.0 +evaluate==0.4.6 +diffusers==0.39.0 +pydantic==2.13.4 +annotated-types==0.8.0 +protobuf==7.35.1 +sentry-sdk==2.66.1 +einops==0.8.2 +packaging==26.2 +nvidia-nvjitlink-cu12==12.4.127 +nvidia-curand-cu12==10.3.5.147 +nvidia-cusparselt-cu12==0.6.2 +nvidia-cusparse-cu12==12.3.1.170 +nvidia-cuda-runtime-cu12==12.4.127 +torchvision==0.21.0+cu124 +nvidia-cuda-nvrtc-cu12==12.4.127 +nvidia-cuda-cupti-cu12==12.4.127 +nvidia-cusolver-cu12==11.6.1.9 +nvidia-cublas-cu12==12.4.5.8 +nvidia-cudnn-cu12==9.1.0.70 +huggingface_hub==1.26.0 +datasets==5.0.1 +safetensors==0.8.0 +accelerate==1.14.0 +pydantic_core==2.46.4 +ninja==1.13.0 +autocommand==2.2.2 +backports.tarfile==1.2.0 +importlib_metadata==8.7.1 +jaraco.text==4.0.0 +jaraco.context==6.1.0 +jaraco.functools==4.4.0 +more-itertools==10.8.0 +packaging==26.0 +platformdirs==4.4.0 +tomli==2.4.0 +wheel==0.46.3 +zipp==3.23.0 diff --git a/wandb/run-20260809_050050-59pftr14/files/wandb-metadata.json b/wandb/run-20260809_050050-59pftr14/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..5a6a19e3e9f60f842855d66a049fedfb3e25db8b --- /dev/null +++ b/wandb/run-20260809_050050-59pftr14/files/wandb-metadata.json @@ -0,0 +1,96 @@ +{ + "os": "Linux-5.15.0-126-generic-x86_64-with-glibc2.35", + "python": "CPython 3.11.15", + "startedAt": "2026-08-09T05:00:50.313572Z", + "args": [ + "--config", + "configs/baseline.yaml", + "--variants", + "glu-linear-94L", + "--push" + ], + "program": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py", + "codePath": "sweep.py", + "codePathLocal": "sweep.py", + "git": { + "remote": "https://github.com/deepnevro/Activation.git", + "commit": "34b8d2e8f9a0c5751333310e69fa0c1056381deb" + }, + "email": "deepnevro@gmail.com", + "root": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation", + "host": "deeplens-k3s-node1", + "executable": "/mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python", + "cpu_count": 112, + "cpu_count_logical": 224, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1560765693952", + "used": "708235583488" + } + }, + "memory": { + "total": "2164089937920" + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-39c684a5-fde6-83d7-1663-0859795881ae" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-132944c4-b689-2b5f-89a4-d730401677ab" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bc6c3e3c-9b90-09ca-c034-774961847c54" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea" + } + ], + "cudaVersion": "12.4", + "writerId": "ou0ddnt36zj23smd7g76a30q6z1cg54s" +} \ No newline at end of file diff --git a/wandb/run-20260809_050050-59pftr14/files/wandb-summary.json b/wandb/run-20260809_050050-59pftr14/files/wandb-summary.json new file mode 100644 index 0000000000000000000000000000000000000000..eaa5220777901eb90db01f1346215aae7aff3458 --- /dev/null +++ b/wandb/run-20260809_050050-59pftr14/files/wandb-summary.json @@ -0,0 +1 @@ +{"train/train/tensor_grad_model_layers_41_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp/std":0.20678761290597703,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/mean":-9.19681042432785e-08,"train/train/tensor_act_model_layers_49_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_embed_tokens/std":0.11572265813892454,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn_q_proj/norm":5170.199303887008,"train/train/tensor_act_model_layers_22_self_attn_o_proj/mean":5.5909156799316406e-05,"train/train/tensor_act_model_layers_73_post_attention_layernorm/norm":5792.613037111685,"train/train/tensor_act_model_layers_50_mlp/std":0.07055707892651966,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/norm":0.02496852681772798,"train/train/layer_model_layers_88/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_mlp_down_proj/max_abs":0.4609375,"train/train/tensor_act_model_layers_28_mlp_gate_proj/mean":-0.00444793701171875,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/max_abs":0.00011920928955078125,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/std":3.0887785575315205e-05,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_gate_proj_weight/std":4.648217783070439e-05,"train/train/tensor_act_model_layers_47/std":1.250000474974423,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/max_abs":0.1171875,"train/train/tensor_act_model_layers_35_self_attn_q_proj/norm":5034.858266150283,"train/train/tensor_act_model_layers_20_input_layernorm/norm":5792.611450197262,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/norm":3.359375,"train/train/tensor_act_model_layers_5_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/mean":0.00013256072998046875,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp/std":0.042542353154980526,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/std":0.050048828125,"train/train/tensor_act_model_layers_91_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_5/act/norm":14437.135208287193,"train/train/tensor_act_model_layers_58_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_gate_proj/mean":-0.0020885467529296875,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/std":7.928639185325247e-05,"train/train/tensor_act_model_layers_73/norm":8761.609703818403,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/max_abs":0.150390625,"train/train/tensor_act_model_layers_11_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_post_attention_layernorm/max_abs":5.65625,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_o_proj/mean":8.660554885864258e-05,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/std":0.05615234375,"train/train/tensor_act_model_layers_56_self_attn_o_proj/max_abs":1.2265625,"train/train/tensor_param_model_layers_30_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_down_proj/mean":0.0004429817199707031,"train/train/tensor_act_model_layers_77_mlp_gate_proj/norm":3983.653228394363,"train/train/tensor_act_model_layers_81_post_attention_layernorm/mean":0.00626373291015625,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/max_abs":0.000591278076171875,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/max_abs":0.000415802001953125,"train/train/tensor_act_model_layers_78_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn_v_proj/std":0.3051770446319059,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/std":6.752377314503314e-05,"train/train/layer_model_layers_70/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn_k_proj/std":0.8242190528254382,"train/train/tensor_param_model_layers_32_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_26_mlp_gate_proj_weight/std":0.0269775390625,"train/train/tensor_act_model_layers_78_self_attn/max_abs":2.234375,"train/train/tensor_act_model_layers_73_input_layernorm/max_abs":5.78125,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/norm":5.3125,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56/norm":7526.505127468818,"train/train/tensor_act_model_layers_3_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/mean":3.600120544433594e-05,"train/train/tensor_act_model_layers_36_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp_up_proj/max_abs":2.046875,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/norm":5.28125,"train/train/tensor_act_model_layers_56_mlp_down_proj/max_abs":0.64453125,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/max_abs":0.000244140625,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/max_abs":0.263671875,"train/train/tensor_act_model_layers_82_self_attn_o_proj/norm":1244.04466746493,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_mlp_gate_proj_weight/max_abs":0.000598907470703125,"train/train/tensor_act_model_layers_57_self_attn_q_proj/mean":0.079833984375,"train/train/tensor_act_model_layers_83_self_attn_q_proj/max_abs":6.5,"train/train/tensor_act_model_layers_23_post_attention_layernorm/std":1.0000015459942788,"train/train/tensor_param_model_layers_30_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/norm":5.625,"train/train/tensor_act_model_layers_47_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_v_proj/std":0.44677843713561444,"train/train/tensor_act_model_layers_84_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40/std":1.2421881492780993,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_v_proj/norm":2650.579533167062,"train/train/layer_model_layers_53/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/mean":-2.5494955480098724e-08,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_58_post_attention_layernorm/std":1.0000006214479868,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/mean":-1.417938619852066e-07,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/mean":4.3440377339720726e-07,"train/train/tensor_act_model_layers_18_self_attn_o_proj/max_abs":0.56640625,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/mean":-9.611248970031738e-07,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/mean":2.0139850676059723e-07,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/std":6.803930525559408e-05,"train/train/tensor_act_model_layers_75_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_o_proj/std":0.07214430087630692,"train/train/tensor_param_model_layers_0_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_q_proj/mean":0.00439453125,"train/train/tensor_act_model_layers_85_self_attn_o_proj/max_abs":2.296875,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/norm":0.01743972120870417,"train/train/tensor_act_model_layers_32_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_input_layernorm/mean":-0.00797271728515625,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/mean":0.0008392333984375,"train/train/tensor_act_model_layers_73_input_layernorm/mean":0.005710601806640625,"train/train/tensor_param_model_layers_7_mlp_gate_proj_weight/mean":-0.000396728515625,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/norm":4.0625,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/max_abs":0.1318359375,"train/train/tensor_act_model_layers_89_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/norm":3.796875,"train/train/layer__model_layers_68/param/std":0.06001704100324208,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/max_abs":0.0888671875,"train/train/tensor_act_model_layers_78_self_attn/frac_near_dtype_limit":0,"eval/loss":1.8066591024398804,"train/train/tensor_act_model_layers_57_self_attn_v_proj/norm":2483.2145151872637,"train/train/tensor_act_model_layers_13/std":1.2812624834778363,"train/train/tensor_act_model_layers_44_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81/max_abs":11.8125,"train/train/layer_model_layers_70/act/max_abs":10.6875,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn_o_proj/mean":-0.00031566619873046875,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/mean":0.0003414154052734375,"train/train/layer_model_layers_8/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_7/act/std":0.6307277771139983,"train/train/tensor_act_model_layers_63/max_abs":10,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/mean":1.2724194675683975e-07,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/mean":4.220008850097656e-05,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/mean":-1.3969838619232178e-07,"train/train/tensor_act_model_layers_7_post_attention_layernorm/norm":5792.609985353542,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/mean":0.00013256072998046875,"train/train/tensor_act_model_layers_72_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/norm":0.019952989104307775,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/max_abs":0.150390625,"train/train/layer_model_layers_46/act/mean":-0.0021862504737717764,"train/train/tensor_param_model_layers_30_mlp_gate_proj_weight/norm":5.09375,"train/train/tensor_act_model_layers_15_self_attn_k_proj/std":0.7949245901553487,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/mean":5.972106009721756e-08,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/std":4.979051644611168e-05,"train/train/layer_model_layers_48/grad/std":4.89211751559145e-05,"train/train/tensor_param_model_layers_56_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/norm":0.0010938806540926435,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/std":6.215784547607681e-05,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/norm":4.78125,"train/train/tensor_act_model_layers_84_mlp_down_proj/std":0.23584148330394997,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/norm":5.15625,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/std":3.989761192393191e-05,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_51/grad/mean":7.721986301975578e-08,"train/train/tensor_act_model_layers_86_mlp_up_proj/std":0.5664062846837363,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_73/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/max_abs":0.1650390625,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/max_abs":0.0002498626708984375,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/norm":0.00395711916781607,"train/train/tensor_grad_model_layers_80_mlp_gate_proj_weight/max_abs":0.00057220458984375,"train/train/tensor_act_model_layers_21_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_gate_proj_weight/mean":-0.00011587142944335938,"train/train/layer_model_layers_27/act/std":0.6332487289082439,"train/train/tensor_param_model_layers_21_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_q_proj/std":0.9169979983949019,"train/train/tensor_act_model_layers_73_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/std":0.06298828125,"train/train/tensor_act_model_layers_69_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_mlp_gate_proj_weight/mean":2.3469328880310059e-07,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/max_abs":0.00011110305786132812,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_23/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_9_mlp_up_proj/max_abs":1.921875,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/norm":4.125,"train/train/tensor_act_model_layers_41_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_down_proj/max_abs":0.51953125,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_79/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_mlp_gate_proj_weight/max_abs":0.000461578369140625,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/std":0.03955078125,"train/train/layer_model_layers_15/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp_down_proj/max_abs":0.341796875,"train/train/layer__model_layers_80/param/max_abs":1,"train/train/tensor_param_model_layers_90_mlp_gate_proj_weight/std":0.05224609375,"train/train/tensor_act_model_layers_39_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/mean":1.0174699127674103e-07,"train/train/tensor_act_model_layers_65_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_q_proj/mean":0.05682373046875,"train/train/tensor_act_model_layers_35_post_attention_layernorm/max_abs":6.6875,"train/train/tensor_act_model_layers_47_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/mean":2.0721927285194397e-08,"train/train/tensor_act_model_layers_18_self_attn_k_proj/std":0.8437515440279356,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/max_abs":0.216796875,"train/train/tensor_act_model_layers_47_post_attention_layernorm/mean":-0.01123046875,"train/train/tensor_grad_model_layers_44_mlp_gate_proj_weight/std":5.0622099686129323e-05,"train/train/tensor_act_model_layers_25_mlp/norm":217.99589836417505,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_o_proj/max_abs":1.2421875,"train/train/tensor_param_model_layers_22_mlp_gate_proj_weight/mean":-6.246566772460938e-05,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/std":5.354916795191476e-05,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/mean":0.00010538101196289062,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/mean":7.946975529193878e-06,"train/train/tensor_act_model_layers_6_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn/norm":491.82890848715755,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/max_abs":0.1767578125,"train/train/tensor_param_model_layers_36_mlp_gate_proj_weight/mean":-4.792213439941406e-05,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/max_abs":0.1484375,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/std":4.5880447248638164e-05,"train/train/tensor_act_model_layers_28_mlp_down_proj/norm":224.18272518682747,"train/train/tensor_act_model_layers_36_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp_up_proj/norm":2231.4602165404744,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/std":0.025390625,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/mean":0.00023937225341796875,"train/train/layer_model_layers_9/act/mean":-0.006366321018763951,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/mean":3.170967102050781e-05,"train/train/layer__model_layers_61/param/mean":0.0015257122736081318,"train/train/tensor_act_model_layers_83_self_attn/norm":1375.3179945246777,"train/train/tensor_act_model_layers_86_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer__model_layers_91/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp_up_proj/max_abs":2.96875,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/mean":-3.1956005841493607e-08,"train/train/tensor_act_model_layers_44_mlp/mean":-0.00022482872009277344,"train/train/tensor_param_model_layers_37_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_49/act/std":0.6278517529057123,"train/train/tensor_act_model_layers_14_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/mean":8.38935375213623e-06,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/mean":-4.291534423828125e-05,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/mean":-1.601874828338623e-07,"train/train/tensor_act_model_layers_52_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/std":4.360281020773992e-05,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/mean":-4.1851308196783066e-08,"train/train/tensor_param_model_layers_16_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp_gate_proj/max_abs":1.9921875,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/std":3.124635437890166e-05,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/max_abs":0.130859375,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/max_abs":0.001190185546875,"train/train/tensor_act_model_layers_42_input_layernorm/norm":5792.610107421921,"train/train/tensor_param_model_layers_19_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_v_proj/norm":2344.257696715214,"train/train/tensor_act_model_layers_0_input_layernorm/max_abs":4.8125,"train/train/tensor_act_model_layers_69_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_post_attention_layernorm/max_abs":5.5,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_83_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/max_abs":0.000576019287109375,"train/train/tensor_act_model_layers_62_mlp_gate_proj/norm":3259.828255230897,"train/train/tensor_act_model_layers_25_mlp_gate_proj/norm":2089.401642714643,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp_gate_proj/std":0.31054703586689664,"train/train/tensor_act_model_layers_69_input_layernorm/norm":5792.6076660190465,"train/train/tensor_act_model_layers_65_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_1/grad/std":0.00010697682086266744,"train/train/tensor_act_model_layers_53_mlp_down_proj/norm":433.7878417626019,"train/train/tensor_grad_model_layers_14_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/norm":6.53125,"train/train/tensor_act_model_layers_16_self_attn_q_proj/mean":-0.01470947265625,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/max_abs":0.00023365020751953125,"train/train/tensor_act_model_layers_67_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_o_proj/std":0.25244304289370284,"train/train/layer__model_layers_85/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_8/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/max_abs":0.134765625,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/mean":8.353963494300842e-07,"train/train/tensor_act_model_layers_92_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_post_attention_layernorm/std":1.000001917940588,"train/train/tensor_act_model_layers_31_mlp/max_abs":0.416015625,"train/train/tensor_grad_model_layers_61_mlp_gate_proj_weight/std":6.951016935111925e-05,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/norm":4.34375,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp_down_proj/max_abs":0.380859375,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/std":5.256569789910094e-05,"train/train/tensor_param_model_layers_83_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_29_self_attn_o_proj/max_abs":0.8984375,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/max_abs":0.275390625,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/norm":0.0158177055754129,"train/train/layer_model_layers_33/grad/std":4.869354425218304e-05,"train/train/tensor_act_model_layers_50_self_attn_k_proj/mean":-0.00286865234375,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/std":0.048095703125,"train/train/tensor_param_model_layers_74_mlp_gate_proj_weight/norm":7.15625,"train/train/tensor_act_model_layers_36_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_input_layernorm/frac_near_user_limit":0,"train_samples_per_second":228.5,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/norm":7.5,"train/train/tensor_param_model_layers_42_mlp_gate_proj_weight/max_abs":0.142578125,"train/train/tensor_act_model_layers_74_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_post_attention_layernorm/std":1.0000004172324264,"train/train/tensor_act_model_layers_39_mlp_down_proj/norm":327.20775240231006,"train/train/tensor_act_model_layers_83_self_attn_q_proj/std":1.1347708972801864,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/norm":5.875,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/max_abs":0.0003070831298828125,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/std":2.8331654435104778e-05,"train/train/tensor_param_model_layers_49_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_83_mlp_down_proj/norm":1273.7648584228398,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/max_abs":0.1552734375,"train/train/layer_model_layers_22/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/mean":-9.629875421524048e-07,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/mean":-6.682239472866058e-07,"train/train/tensor_act_model_layers_77_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44/norm":7203.681101434,"train/train/tensor_act_model_layers_59_mlp_gate_proj/std":0.4003907320937765,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/max_abs":0.0002727508544921875,"train/train/tensor_act_model_layers_16_self_attn_v_proj/norm":1747.1719293138215,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/std":6.715267469427656e-05,"train/train/tensor_act_model_layers_51/frac_near_dtype_limit":0,"train/train/layer_model_layers_13/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/max_abs":0.000614166259765625,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/norm":0.0006680296283162343,"train/train/tensor_act_model_layers_7_self_attn_k_proj/mean":0.021514892578125,"train/train/tensor_act_model_layers_2_mlp/std":0.18408398475419915,"train/train/tensor_act_model_layers_64_mlp_down_proj/mean":0.0005869865417480469,"train/train/tensor_grad_model_layers_50_mlp_gate_proj_weight/max_abs":0.00054168701171875,"train/train/tensor_act_model_layers_93_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_mlp_gate_proj/norm":1951.8558697581175,"train/train/tensor_act_model_layers_35_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn_v_proj/mean":-0.0011053085327148438,"train/train/tensor_param_model_layers_74_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_q_proj/max_abs":5.53125,"train/train/layer_model_layers_24/act/max_abs":8.5625,"train/train/tensor_act_model_layers_65_self_attn_v_proj/std":0.45947345004889806,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_81/act/mean":0.01798467125211443,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_17_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/std":6.409741654555019e-05,"train/train/layer__model_layers_64/param/std":0.05635699234875925,"train/train/tensor_act_model_layers_21_self_attn_k_proj/std":0.7851569522076404,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/mean":0.0003414154052734375,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/norm":2.671875,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/max_abs":0.0022735595703125,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/max_abs":0.001312255859375,"train/train/tensor_param_model_layers_83_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/mean":-8.106231689453125e-05,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/std":0.0001330902125015825,"train/train/tensor_act_model_layers_4_self_attn_v_proj/norm":2067.395571360583,"train/train/tensor_param_model_layers_29_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn_q_proj/std":1.0000001248845285,"train/train/tensor_act_model_layers_1_mlp/std":0.2285166778856051,"train/train/tensor_act_model_layers_12_mlp_up_proj/mean":0.00079345703125,"train/train/tensor_param_model_layers_2_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/mean":-8.20159912109375e-05,"train/train/tensor_act_model_layers_81_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/std":6.0391666180772674e-05,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/norm":0.019441493743337836,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57/mean":-0.0011456012725830078,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/mean":0.0007476806640625,"train/train/layer_model_layers_2/act/norm":14821.029711535988,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/norm":0.01767650202340647,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/std":0.036376953125,"train/train/tensor_act_model_layers_65_mlp_up_proj/max_abs":2.515625,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/mean":2.8133392333984375e-05,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/max_abs":0.1455078125,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/mean":4.168599843978882e-06,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/max_abs":0.000537872314453125,"train/train/tensor_act_model_layers_85_self_attn/max_abs":2.296875,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/mean":0.00022029876708984375,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_k_proj/std":0.8437500596046428,"train/train/tensor_param_model_layers_57_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_71/grad/max_abs":0.0015869140625,"train/train/tensor_act_model_layers_31_self_attn_v_proj/max_abs":2.0625,"train/train/tensor_act_model_layers_88_self_attn_k_proj/max_abs":5.40625,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_input_layernorm/mean":-0.0139007568359375,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_post_attention_layernorm/std":1.0000016683465853,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/std":4.478890998624001e-05,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_v_proj/norm":1356.4384356052726,"train/train/tensor_act_model_layers_1_input_layernorm/mean":-0.02398681640625,"train/train/tensor_act_model_layers_54_mlp_up_proj/norm":3008.3546378461583,"train/train/tensor_act_model_layers_44_self_attn_o_proj/norm":459.51288475145,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/max_abs":0.2353515625,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/norm":4.25,"train/train/layer__model_layers_36/param/norm":21.665109118695895,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_90_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_v_proj/norm":2385.3914518045704,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/norm":0.011973862579458412,"train/train/layer_model_layers_81/act/norm":17908.411573315774,"train/train/tensor_act_model_layers_78_self_attn_q_proj/std":1.1015627853413477,"train/train/tensor_act_model_layers_83_mlp_gate_proj/max_abs":3.3125,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/max_abs":0.000644683837890625,"train/train/layer__model_layers_73/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_mlp_up_proj/mean":-0.0054168701171875,"train/train/layer__model_layers_18/param/std":0.04965950432390529,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/max_abs":0.1611328125,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/std":3.066131853853975e-05,"train/train/tensor_act_model_layers_72_self_attn/norm":896.9917960355883,"train/train/global/grad/mean":-1.67887470446556e-08,"train/train/tensor_param_model_layers_42_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_gate_proj/max_abs":2.25,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_4_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/std":0.03759765625,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/norm":0.017015020879945098,"train/train/tensor_act_model_layers_10_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/max_abs":0.0003414154052734375,"train/train/tensor_act_model_layers_60_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/mean":5.73229044675827e-07,"train/train/tensor_act_model_layers_79_input_layernorm/norm":5792.608764651518,"train/train/tensor_act_model_layers_91_mlp_down_proj/mean":0.0019741058349609375,"train/train/tensor_act_model_layers_53/norm":7398.065234096585,"train/train/tensor_act_model_layers_17_mlp_gate_proj/std":0.24975623223148968,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/norm":6,"train/train/tensor_act_model_layers_67_self_attn_q_proj/mean":-0.019744873046875,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_8_post_attention_layernorm/max_abs":5.25,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/norm":0.026525586789765784,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/std":2.5055747066237902e-05,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/mean":-0.000202178955078125,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/std":3.891858307890582e-05,"train/train/tensor_grad_model_layers_7_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/max_abs":0.00028228759765625,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/std":1.947578232661653e-05,"train/train/tensor_param_model_layers_15_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/mean":2.6938505470752716e-07,"train/train/tensor_act_model_layers_49_mlp_gate_proj/max_abs":1.9609375,"train/train/tensor_act_model_layers_84_self_attn_k_proj/norm":6140.353553228541,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn/norm":1312.0046012674302,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/std":8.645337966035832e-05,"train/train/tensor_act_model_layers_5_mlp/frac_near_dtype_limit":0,"train/train/global/grad/max_abs":0.0128173828125,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_mlp_gate_proj_weight/max_abs":0.000530242919921875,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_73_mlp_down_proj/mean":0.0010747909545898438,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/std":0.046630859375,"train/train/tensor_param_model_layers_53_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/norm":0.02826514296492316,"train/train/tensor_act_model_layers_57_input_layernorm/mean":-0.00168609619140625,"train/train/tensor_param_model_layers_2_mlp_gate_proj_weight/mean":2.205371856689453e-05,"train/train/tensor_grad_model_layers_8_mlp_gate_proj_weight/std":4.3559319631710904e-05,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/std":4.484861597824961e-05,"train/train/tensor_act_model_layers_72_self_attn_q_proj/max_abs":8.25,"train/train/tensor_act_model_layers_1_mlp_gate_proj/std":0.35546875360247854,"train/train/tensor_act_model_layers_17_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/norm":0.013685151874192884,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/max_abs":0.0006866455078125,"train/train/tensor_act_model_layers_53_self_attn_v_proj/max_abs":2.703125,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/mean":0.00022983551025390625,"train/train/tensor_act_model_layers_25_mlp/max_abs":0.390625,"train/train/layer__model_layers_61/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_mlp_gate_proj_weight/mean":9.5367431640625e-05,"train/train/time_per_step_avg":1.8038726660981774,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/mean":-0.00011444091796875,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/std":0.048828125,"train/train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn_v_proj/std":0.3964845020452246,"train/train/layer_model_layers_88/act/mean":0.012974602835518973,"train/train/tensor_act_model_layers_22_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/std":0.045654296875,"train/train/tensor_act_model_layers_27_mlp_gate_proj/max_abs":2.015625,"train/train/tensor_act_model_layers_68_self_attn_o_proj/norm":1144.5357810048406,"train/train/tensor_act_model_layers_35/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp_gate_proj/mean":-0.004199981689453125,"train/train/tensor_act_model_layers_78_mlp_down_proj/mean":0.0010099411010742188,"train/train/layer_model_layers_7/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/max_abs":0.001220703125,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_11/act/mean":-0.005650392600468227,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/max_abs":0.000522613525390625,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/std":7.134399228153709e-05,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/norm":5.15625,"train/train/tensor_act_model_layers_39_mlp_gate_proj/mean":-8.821487426757812e-06,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_88_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_mlp_up_proj/max_abs":1.8515625,"train/train/tensor_act_model_layers_84_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_post_attention_layernorm/max_abs":6.09375,"train/train/tensor_act_model_layers_79_mlp/norm":1050.8473044482769,"train/train/tensor_act_model_layers_75_mlp/norm":904.0612064637012,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/mean":-2.557062543928623e-07,"train/train/tensor_act_model_layers_56_self_attn_o_proj/std":0.11096270598815598,"train/train/tensor_act_model_layers_78_self_attn_k_proj/max_abs":5.78125,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_o_proj/max_abs":1.6953125,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/std":4.1326320547600064e-05,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/max_abs":0.2060546875,"train/train/tensor_act_model_layers_36_mlp/std":0.05053712073758405,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/max_abs":0.162109375,"train/train/layer_model_layers_74/act/std":0.7444705533437278,"train/train/tensor_act_model_layers_86/max_abs":12.125,"train/train/layer__model_layers_2/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_mlp_gate_proj_weight/norm":4.71875,"train/train/tensor_act_model_layers_73_post_attention_layernorm/mean":0.006443023681640625,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/norm":0.0026284897397791025,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_49/param/mean":0.0014860299001803823,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/norm":0.08139518091431593,"train/train/tensor_act_model_layers_50_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn_k_proj/std":1.0625003470271697,"train/train/tensor_act_model_layers_5_input_layernorm/max_abs":4.9375,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/norm":0.0032867109320148385,"train/train/tensor_act_model_layers_49_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/norm":0.024287697877337967,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/norm":0.028601011698675266,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_60/act/norm":14040.142817360515,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/norm":0.025455538593694343,"train/train/tensor_act_model_layers_1_mlp_up_proj/mean":-0.01776123046875,"train/train/layer__model_layers_23/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp/mean":0.00014793872833251953,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/max_abs":0.10595703125,"train/train/layer__model_layers_29/param/std":0.05074712009608383,"train/train/tensor_act_model_layers_17_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_post_attention_layernorm/max_abs":6.0625,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/mean":1.3455748558044434e-05,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp_up_proj/mean":0.00040662288665771484,"train/train/tensor_act_model_layers_89_mlp_up_proj/std":0.6328125533958253,"train/train/layer__model_layers_33/param/max_abs":1,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/norm":0.018167265742891254,"train/train/tensor_act_model_layers_24_mlp_up_proj/norm":2056.270924989413,"train/train/tensor_param_model_layers_70_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_29_mlp_gate_proj_weight/std":0.02783203125,"train/train/tensor_act_model_layers_3_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/mean":-0.0843505859375,"train/train/layer__model_layers_11/param/mean":0.0017084130630850233,"train/train/tensor_act_model_layers_81_post_attention_layernorm/norm":5792.609008790157,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/mean":5.602836608886719e-06,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/norm":8.375,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn/mean":-0.00018978118896484375,"train/train/tensor_act_model_layers_56_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/norm":0.012347632352907829,"train/train/layer__model_layers_31/param/norm":20.82666559965853,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/max_abs":0.255859375,"train/train/tensor_param_model_layers_77_mlp_gate_proj_weight/max_abs":0.171875,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_post_attention_layernorm/max_abs":6.71875,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_k_proj/mean":0.0423583984375,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/mean":-2.6694033294916153e-07,"train/train/tensor_act_model_layers_33_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/mean":0.00011539459228515625,"train/train/tensor_act_model_layers_22_mlp_gate_proj/mean":-0.002735137939453125,"train/train/tensor_act_model_layers_69_self_attn_q_proj/mean":-0.0066680908203125,"train/train/tensor_act_model_layers_28_self_attn_v_proj/std":0.4282236790257434,"train/train/tensor_act_model_layers_51_self_attn_q_proj/norm":4919.809117730293,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_20/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/mean":6.4849853515625e-05,"train/train/tensor_act_model_layers_20/norm":7342.532806397416,"train/train/tensor_act_model_layers_33_mlp_down_proj/mean":0.0005474090576171875,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/std":0.024658203125,"train/train/tensor_act_model_layers_0_mlp_gate_proj/std":0.777343751797125,"train/train/tensor_param_model_layers_61_mlp_gate_proj_weight/norm":6.625,"train/train/tensor_act_model_layers_6_mlp_up_proj/std":0.23584036994859586,"train/train/tensor_grad_model_layers_20_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/std":2.57551941073011e-05,"train/train/tensor_act_model_layers_38_mlp/mean":0.0003247261047363281,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/max_abs":0.19140625,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/max_abs":0.000652313232421875,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/max_abs":0.000606536865234375,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/std":9.260480810296449e-05,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_input_layernorm/std":1.0000003416498073,"train/train/tensor_act_model_layers_75_self_attn_k_proj/std":0.9160177637430592,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/std":5.661644435051682e-05,"train/train/tensor_act_model_layers_5_mlp_down_proj/max_abs":0.71484375,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/norm":6.75,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/std":0.054443359375,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/mean":-2.777576446533203e-05,"train/train/tensor_act_model_layers_41_input_layernorm/std":1.0000004566972032,"train/train/tensor_act_model_layers_14_self_attn_q_proj/max_abs":5.53125,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_23/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_gate_proj_weight/max_abs":0.1767578125,"train/train/layer_model_layers_63/act/max_abs":10,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/mean":6.866455078125e-05,"train/train/tensor_grad_model_layers_24_mlp_gate_proj_weight/norm":0.014549781457744325,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_86/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn/std":0.1129172474534032,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/mean":2.430751919746399e-07,"train/train/tensor_act_model_layers_57_mlp_gate_proj/max_abs":2.21875,"train/train/tensor_act_model_layers_69_self_attn_v_proj/norm":2402.9089404245824,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/max_abs":0.12451171875,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/max_abs":0.000885009765625,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/norm":0.0017763468676721364,"train/train/layer__model_layers_26/param/norm":20.557114843847252,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_q_proj/norm":5734.834257203457,"train/train/tensor_act_model_layers_26_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/norm":6.3125,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/norm":0.01596877292922552,"train/train/tensor_act_model_layers_55_mlp_down_proj/mean":0.0002727508544921875,"train/train/tensor_param_model_layers_18_mlp_gate_proj_weight/std":0.026123046875,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/std":0.057861328125,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70/mean":0.007991790771484375,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/max_abs":0.248046875,"train/train/tensor_act_model_layers_69_mlp_up_proj/std":0.4453125554218592,"train/train/layer_model_layers_73/grad/max_abs":0.00165557861328125,"train/train/tensor_act_model_layers_24/mean":-0.02349853515625,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_90/act/max_abs":13.25,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_19/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/mean":-0.0003604888916015625,"train/train/tensor_act_model_layers_25_self_attn_v_proj/norm":2076.708728110487,"train/train/tensor_act_model_layers_76_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/max_abs":0.000179290771484375,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/mean":-1.118052750825882e-06,"train/train/layer__model_layers_36/param/mean":0.0015462556978841655,"train/train/tensor_param_model_layers_3_mlp_gate_proj_weight/std":0.0233154296875,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/mean":-1.0073184967041016e-05,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/mean":3.6670826375484467e-09,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_gate_proj_weight/norm":6,"train/train/tensor_param_model_layers_58_mlp_gate_proj_weight/mean":2.8014183044433594e-06,"train/train/tensor_param_model_layers_89_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/mean":-6.723403930664062e-05,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/norm":0.0016612732859829797,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/max_abs":0.2294921875,"train/train/tensor_act_model_layers_22_mlp_down_proj/norm":210.19898898132539,"train/train/layer_model_layers_87/act/max_abs":11.875,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/norm":0.02753720649254598,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/max_abs":0.001068115234375,"train/train/tensor_act_model_layers_45_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/mean":7.361173629760742e-06,"train/train/tensor_act_model_layers_70_self_attn_q_proj/std":0.9980493687105845,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/norm":0.007578519953434549,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/norm":0.019380464244760042,"train/train/tensor_act_model_layers_49_self_attn_v_proj/mean":0.00481414794921875,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_72/param/max_abs":1,"train/train/layer_model_layers_18/act/norm":13408.227328448189,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/mean":0.0003070831298828125,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/std":6.57987069740013e-05,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/norm":0.0009189304047570197,"train/train/tensor_act_model_layers_48_self_attn_q_proj/norm":5096.262059059395,"train/train/tensor_act_model_layers_27_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/max_abs":0.000308990478515625,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/std":3.1125108765071484e-05,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/std":0.0242919921875,"train/train/tensor_act_model_layers_92_self_attn/max_abs":2.640625,"train/train/tensor_act_model_layers_4_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn/mean":-0.0005841255187988281,"train/train/layer__model_layers_48/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_q_proj/mean":0.0626220703125,"train/train/tensor_act_model_layers_58_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_q_proj/max_abs":6.125,"train/train/layer__model_layers_74/param/mean":0.0015088347674531981,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp_gate_proj/std":0.24877967225529368,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/max_abs":0.000202178955078125,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/std":2.7562350221276535e-05,"train/train/tensor_act_model_layers_45_self_attn_k_proj/max_abs":4.6875,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/mean":-0.000713348388671875,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_27/grad/std":4.551347853208744e-05,"train/train/tensor_act_model_layers_14_input_layernorm/mean":-0.029754638671875,"train/train/tensor_act_model_layers_87_mlp_up_proj/std":0.6015629063951681,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/mean":2.7418136596679688e-05,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/norm":0.029275886646985307,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/norm":0.025701091883059673,"train/train/tensor_act_model_layers_62_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/mean":3.186287358403206e-07,"train/train/tensor_act_model_layers_84_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp/norm":242.92867394544308,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/std":0.060546875,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/mean":-0.0002002716064453125,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/mean":-6.3800252974033356e-06,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/norm":6.03125,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/max_abs":0.0001239776611328125,"train/train/tensor_act_model_layers_52_self_attn_k_proj/std":0.6933626309878964,"train/train/tensor_act_model_layers_2_mlp_up_proj/std":0.32617189891323983,"train/train/tensor_param_model_layers_52_mlp_gate_proj_weight/std":0.033203125,"train/train/tensor_act_model_layers_62_mlp_up_proj/norm":3290.905236267858,"train/train/tensor_act_model_layers_29_input_layernorm/max_abs":6.5,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/max_abs":0.00015544891357421875,"train/train/tensor_act_model_layers_47_post_attention_layernorm/norm":5792.60437012069,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/std":6.760066137200854e-05,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/mean":-9.965896606445312e-05,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/max_abs":0.000576019287109375,"train/train/tensor_param_model_layers_30_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_q_proj/max_abs":4.6875,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/norm":3.046875,"train/train/tensor_act_model_layers_13_self_attn_q_proj/mean":-0.025238037109375,"train/train/layer_model_layers_66/act/std":0.6926072172076951,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/mean":-0.0001163482666015625,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/mean":0.000133514404296875,"train/train/tensor_param_model_layers_16_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/std":3.942206992946557e-05,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn_v_proj/std":0.2597656603301929,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_mlp_gate_proj_weight/max_abs":0.1240234375,"train/train/tensor_act_model_layers_42_self_attn_q_proj/max_abs":5.71875,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_10_mlp_gate_proj/std":0.2316898454460198,"train/train/tensor_act_model_layers_52/norm":7363.122827411051,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/norm":6.5625,"train/train/tensor_act_model_layers_89_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/max_abs":0.0004825592041015625,"train/train/tensor_act_model_layers_88_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp/norm":239.01728178106117,"train/train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn_v_proj/norm":1656.4952793733396,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/norm":0.00147747514476803,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/norm":0.02286831820201,"train/train/tensor_act_model_layers_43_self_attn_q_proj/std":0.9345718979697496,"train/train/tensor_act_model_layers_30_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_o_proj/mean":-0.0004100799560546875,"train/train/tensor_act_model_layers_20_self_attn_q_proj/norm":4675.7492880054615,"train/train/tensor_act_model_layers_92_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_q_proj/mean":-0.023345947265625,"train/train/tensor_grad_model_layers_14_mlp_gate_proj_weight/max_abs":0.0003662109375,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/norm":0.008797641381983808,"train/train/tensor_act_model_layers_56_self_attn_o_proj/mean":-0.0003657341003417969,"train/train/tensor_act_model_layers_30_input_layernorm/max_abs":6.59375,"train/train/tensor_act_model_layers_23_self_attn_k_proj/max_abs":4.40625,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/max_abs":0.00026702880859375,"train/train/tensor_act_model_layers_12_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/mean":0.00038909912109375,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/std":0.037109375,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_up_proj/norm":2136.2609884844874,"train/train/tensor_grad_model_layers_44_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/norm":0.003198400158657175,"train/train/tensor_grad_model_layers_36_mlp_gate_proj_weight/max_abs":0.00042724609375,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/std":9.59921404426044e-05,"train/train/tensor_act_model_layers_53_self_attn_k_proj/mean":-0.019805908203125,"train/train/tensor_act_model_layers_79_self_attn_o_proj/norm":809.8521964442652,"train/train/layer_model_layers_15/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_k_proj/mean":-0.06048583984375,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_q_proj/std":1.0410212231247558,"train/train/tensor_param_model_layers_73_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_31_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/max_abs":0.00013828277587890625,"train/train/tensor_act_model_layers_23_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp/norm":357.7946490354836,"train/train/tensor_act_model_layers_79_input_layernorm/std":1.0000012480878864,"train/train/layer_model_layers_19/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_input_layernorm/max_abs":6.40625,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/norm":4.09375,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/max_abs":0.158203125,"train/train/tensor_act_model_layers_42_mlp_gate_proj/mean":-0.0082244873046875,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_28/act/max_abs":9,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/std":0.02587890625,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/std":0.0439453125,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/max_abs":0.00103759765625,"train/train/tensor_act_model_layers_46_self_attn_o_proj/norm":557.2084066712753,"train/train/tensor_act_model_layers_51_self_attn_v_proj/mean":-0.00036144256591796875,"train/train/tensor_act_model_layers_87_self_attn_o_proj/max_abs":2.375,"train/train/tensor_act_model_layers_28_input_layernorm/mean":-0.013580322265625,"train/train/tensor_act_model_layers_51_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_71_self_attn/norm":836.6121965638475,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/std":0.0244140625,"train/train/tensor_act_model_layers_26_input_layernorm/max_abs":6.4375,"train/train/tensor_act_model_layers_15_self_attn_v_proj/mean":-0.00171661376953125,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/std":0.038330078125,"train/train/tensor_act_model_layers_65_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_mlp_gate_proj_weight/max_abs":0.000560760498046875,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/mean":-7.888302206993103e-07,"train/train/tensor_act_model_layers_64_mlp_up_proj/mean":0.00038635730743408203,"train/train/tensor_act_model_layers_46_mlp_down_proj/max_abs":0.474609375,"train/train/tensor_act_model_layers_46_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/max_abs":0.00021648406982421875,"train/train/tensor_act_model_layers_11_self_attn_q_proj/std":0.9052750876534081,"train/train/layer_model_layers_39/act/std":0.6426059409555154,"train/train/tensor_act_model_layers_45_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_16/grad/mean":1.215528516613191e-07,"train/train/layer_model_layers_81/grad/norm":0.0737202840403852,"train/train/layer_model_layers_59/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_mlp_gate_proj_weight/std":9.21872200487928e-05,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/norm":4.84375,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/std":0.051513671875,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/std":4.816654642048769e-05,"train/train/tensor_act_model_layers_76_mlp_gate_proj/mean":0.0065460205078125,"train/train/tensor_act_model_layers_11_mlp/mean":-6.955862045288086e-05,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/mean":-1.280568540096283e-09,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_22_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/max_abs":5.125,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/max_abs":0.22265625,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/mean":0.000568389892578125,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp_gate_proj/mean":-0.00037360191345214844,"train/train/tensor_act_model_layers_72_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/std":7.165794837933934e-05,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/std":4.8018592164095726e-05,"train/train/layer__model_layers_56/param/mean":0.0014465141593945007,"train/train/layer__model_layers_47/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_gate_proj_weight/max_abs":0.0006103515625,"train/train/tensor_grad_model_layers_32_mlp_gate_proj_weight/mean":1.327134668827057e-07,"train/train/tensor_act_model_layers_55_post_attention_layernorm/std":1.0000002976593816,"train/train/tensor_act_model_layers_71_self_attn_k_proj/max_abs":5.40625,"train/train/tensor_act_model_layers_66_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/mean":5.61494380235672e-06,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/std":0.0498046875,"train/train/layer_model_layers_79/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57/max_abs":9.9375,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/norm":7.84375,"train/train/tensor_act_model_layers_45_mlp_gate_proj/max_abs":2.140625,"train/train/tensor_act_model_layers_93_self_attn_q_proj/mean":-0.0333251953125,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/max_abs":0.000255584716796875,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/norm":0.014075519574706674,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_4_self_attn_k_proj/norm":7283.993806330314,"train/train/layer_model_layers_47/act/norm":13682.858351636212,"train/train/tensor_act_model_layers_11_input_layernorm/max_abs":5.625,"train/train/tensor_act_model_layers_54/mean":-0.0035858154296875,"train/train/tensor_act_model_layers_10/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_k_proj/mean":0.0139312744140625,"train/train/layer_model_layers_10/act/max_abs":8.8125,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/norm":7.1875,"train/train/layer_model_layers_74/grad/std":7.771552493838235e-05,"train/train/tensor_act_model_layers_93_self_attn_v_proj/norm":2840.076652244117,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_80/grad/mean":-2.7216994167304077e-07,"train/train/tensor_act_model_layers_18_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/max_abs":0.0007476806640625,"train/train/tensor_act_model_layers_91_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_15/grad/std":4.544181684480882e-05,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/max_abs":0.1669921875,"train/train/tensor_act_model_layers_20_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_70/grad/max_abs":0.00174713134765625,"train/train/tensor_act_model_layers_8_mlp/max_abs":0.6015625,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/norm":0.008033676364462394,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/mean":1.7434358596801758e-05,"train/train/tensor_act_model_layers_41/max_abs":9.375,"train/train/tensor_act_model_layers_31_mlp/std":0.04425063116774016,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/norm":5.5625,"train/train/tensor_act_model_layers_65_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/norm":0.01816870107181602,"train/train/tensor_act_model_layers_81_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_mlp_up_proj/max_abs":2.875,"train/train/layer_model_layers_2/grad/norm":0.0751588006529557,"train/train/tensor_act_model_layers_22_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/max_abs":3.359375,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_up_proj/norm":3355.4932150116388,"train/train/layer_model_layers_6/grad/norm":0.049049446048353346,"train/train/tensor_act_model_layers_32_self_attn_v_proj/std":0.36328136540386713,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_v_proj/std":0.35888781808401754,"train/train/layer_model_layers_92/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/std":0.0242919921875,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/mean":3.3830292522907257e-07,"train/train/tensor_act_model_layers_82_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_post_attention_layernorm/std":1.0000014811219389,"train/train/tensor_param_model_layers_74_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/mean":-7.508788257837296e-09,"train/train/layer_model_layers_48/grad/max_abs":0.00115966796875,"train/train/tensor_act_model_layers_47_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/max_abs":5.34375,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/norm":5375.258676170626,"train/train/tensor_act_model_layers_8_self_attn_o_proj/max_abs":0.53125,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/norm":4.46875,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp_up_proj/max_abs":1.953125,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/norm":5.03125,"train/train/tensor_act_model_layers_13/frac_near_user_limit":0,"train/train/layer__model_layers_10/param/norm":19.871891954309483,"train/train/layer__model_layers_76/param/norm":23.92484979054205,"train/train/tensor_act_model_layers_75_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_6/act/norm":14319.770156415982,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/max_abs":0.000308990478515625,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/std":3.153779980927055e-05,"train/train/tensor_act_model_layers_21_mlp_down_proj/norm":238.57130885392115,"train/train/tensor_act_model_layers_36_mlp_down_proj/std":0.05053712073758405,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_12/param/mean":0.001536448176677067,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/norm":0.028907154687694368,"train/train/tensor_act_model_layers_12/std":1.287114380695062,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/std":1.827457887013756e-05,"train/train/tensor_act_model_layers_64_self_attn_o_proj/max_abs":1.3984375,"train/train/tensor_act_model_layers_72_self_attn_o_proj/mean":-0.0009145736694335938,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/norm":0.015739172795906288,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/mean":-5.23589551448822e-06,"train/train/tensor_act_model_layers_75_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp/max_abs":0.435546875,"train/train/tensor_param_model_layers_47_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_90_self_attn/max_abs":3.28125,"train/train/layer_model_layers_61/act/norm":14406.156872077916,"train/train/tensor_act_model_layers_74/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/mean":0.0006561279296875,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/mean":6.295740604400635e-07,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/norm":6.3125,"train/train/tensor_act_model_layers_20_self_attn_v_proj/mean":-0.00020191073417663574,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/norm":6.59375,"train/train/layer_model_layers_89/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/max_abs":0.000701904296875,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/std":0.048583984375,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/std":0.02392578125,"train/train/tensor_act_model_layers_10/norm":7453.845297648927,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_v_proj/max_abs":2,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/mean":3.2815150916576385e-06,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/std":0.0306396484375,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/max_abs":0.0002803802490234375,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/std":7.673040697626602e-05,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/max_abs":0.109375,"train/train/tensor_grad_model_layers_80_mlp_gate_proj_weight/std":8.011948237465142e-05,"train/train/tensor_act_model_layers_69_mlp_down_proj/std":0.13476564047817538,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/max_abs":0.00124359130859375,"train/train/layer__model_layers_8/param/mean":0.0014446573956708268,"train/train/layer_model_layers_9/grad/std":5.2008968682763005e-05,"train/train/tensor_act_model_layers_84_mlp_gate_proj/norm":4457.486243630319,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/max_abs":0.0004329681396484375,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/mean":-2.2142194211483002e-07,"train/train/tensor_param_model_layers_87_mlp_gate_proj_weight/max_abs":0.2255859375,"train/train/tensor_param_model_layers_64_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_26_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/norm":0.010414769079301101,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/norm":3.140625,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/std":4.932178396463866e-05,"train/train/layer__model_layers_59/param/max_abs":1,"train/train/tensor_act_model_layers_42_mlp_down_proj/norm":345.73317908750266,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/max_abs":0.00077056884765625,"train/train/tensor_act_model_layers_6_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn_o_proj/mean":0.0012416839599609375,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/mean":0.0004138946533203125,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/max_abs":0.1572265625,"train/train/tensor_param_model_layers_20_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/max_abs":0.000640869140625,"train/train/tensor_act_model_layers_65_self_attn_o_proj/max_abs":1.9765625,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_51/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/max_abs":1,"train/train/layer__model_layers_47/param/max_abs":1,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_mlp_up_proj/max_abs":2.15625,"train/train/tensor_act_model_layers_51_mlp_up_proj/mean":-0.00762176513671875,"train/train/tensor_act_model_layers_1_post_attention_layernorm/mean":-0.025146484375,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_o_proj/max_abs":1.453125,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_mlp_gate_proj_weight/mean":1.5739351511001587e-07,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_mlp_down_proj/norm":925.7319424835587,"train/train/tensor_act_model_layers_41_mlp/mean":0.0005826950073242188,"train/train/tensor_act_model_layers_41_self_attn_o_proj/mean":-0.0003724396228790283,"train/train/tensor_act_model_layers_69/max_abs":10.8125,"train/train/tensor_param_model_layers_56_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/mean":-1.1490192264318466e-07,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/norm":0.007470420133927945,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_mlp_gate_proj_weight/max_abs":0.00048065185546875,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/mean":-0.00022602081298828125,"train/train/tensor_param_model_layers_20_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_36_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/norm":0.022200442152732223,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/mean":3.180466592311859e-07,"train/train/tensor_param_model_layers_5_mlp_gate_proj_weight/mean":-0.00013446807861328125,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/max_abs":0.0003490447998046875,"train/train/tensor_act_model_layers_48_self_attn_v_proj/mean":0.0016994476318359375,"train/train/tensor_param_model_layers_92_mlp_gate_proj_weight/max_abs":0.244140625,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/max_abs":0.251953125,"train/train/tensor_act_model_layers_53_mlp/max_abs":0.53125,"train/train/tensor_act_model_layers_10_input_layernorm/max_abs":5.46875,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/std":0.041015625,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn_k_proj/mean":0.018768310546875,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/max_abs":0.1259765625,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/std":0.00011869493548498461,"train/train/tensor_act_model_layers_90_self_attn_k_proj/max_abs":6.125,"train/train/tensor_act_model_layers_27_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_q_proj/norm":4847.962710779017,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/norm":0.0007870026479402584,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/norm":6.59375,"train/train/tensor_act_model_layers_20_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_k_proj/mean":-0.0088348388671875,"train/train/tensor_act_model_layers_36_mlp/max_abs":0.4609375,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_post_attention_layernorm/max_abs":6.125,"train/train/tensor_act_model_layers_81_mlp_down_proj/max_abs":1.53125,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_k_proj/max_abs":5.4375,"train/train/layer_model_layers_15/act/std":0.6311689442662832,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/mean":5.25033101439476e-08,"train/train/tensor_act_model_layers_30_self_attn/std":0.07214430087630692,"train/train/layer__model_layers_15/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_35/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/mean":-4.772446118295193e-07,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/mean":4.366040229797363e-06,"train/train/tensor_act_model_layers_80_mlp_down_proj/max_abs":1.4453125,"train/train/tensor_param_model_layers_61_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/std":0.00013811655609064283,"train/train/tensor_act_model_layers_6_self_attn_k_proj/std":0.9326230328561433,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_mlp_gate_proj_weight/mean":0.0001583099365234375,"train/train/layer__model_layers_15/param/std":0.05010154926764521,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/mean":-1.7345882952213287e-08,"train/train/tensor_act_model_layers_29_self_attn/norm":469.32216258012346,"train/train/tensor_act_model_layers_30_self_attn_v_proj/norm":1908.7541116534676,"train/train/tensor_act_model_layers_86_self_attn_k_proj/std":1.0293025843626629,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/norm":7.9375,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/std":0.0263671875,"train/train/tensor_act_model_layers_89_self_attn/std":0.17675783966957934,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/std":5.487439408640119e-05,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/std":0.0277099609375,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/std":5.176120945439516e-05,"train/train/tensor_param_model_layers_29_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/std":0.0625,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/mean":0.000492095947265625,"train/train/tensor_grad_model_layers_60_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/norm":0.00440503560402832,"train/train/tensor_act_model_layers_22_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp_gate_proj/max_abs":4.96875,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/norm":0.03220488079009893,"train/train/tensor_act_model_layers_84_self_attn_q_proj/max_abs":6.46875,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/max_abs":0.00058746337890625,"train/train/layer__model_layers_43/param/norm":21.732100213451528,"train/train/layer__model_layers_22/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_post_attention_layernorm/max_abs":4.96875,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/max_abs":0.0001068115234375,"train/train/layer_model_layers_88/grad/max_abs":0.00115966796875,"train/train/tensor_act_model_layers_93_mlp_down_proj/max_abs":11.75,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_q_proj/mean":0.05108642578125,"train/train/tensor_act_model_layers_11_post_attention_layernorm/norm":5792.610961922535,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/max_abs":0.0004711151123046875,"train/train/layer_model_layers_14/grad/std":5.0027594596573726e-05,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/max_abs":0.125,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/mean":4.572211764752865e-08,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/std":0.0257568359375,"train/train/tensor_act_model_layers_62_self_attn/mean":0.00039768218994140625,"train/train/tensor_param_model_layers_39_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/norm":4.9375,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/norm":0.0008231529801795324,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/std":0.058349609375,"train/train/tensor_act_model_layers_5_input_layernorm/std":1.000000094994898,"train/train/layer__model_layers_16/param/std":0.0500413521767147,"train/train/tensor_act_model_layers_5_self_attn_o_proj/std":0.06884864042635273,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/max_abs":0.19921875,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_mlp_gate_proj_weight/max_abs":0.1611328125,"train/train/tensor_act_model_layers_87_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/max_abs":0.00023174285888671875,"train/train/tensor_act_model_layers_39_input_layernorm/mean":-0.012054443359375,"train/train/tensor_act_model_layers_13_self_attn/std":0.05676313035664301,"train/train/layer__model_layers_69/param/mean":0.0015400664854719188,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/mean":-8.866190910339355e-07,"train/train/tensor_act_model_layers_45_post_attention_layernorm/std":1.0000003566964824,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_20_mlp_up_proj/mean":-0.00150299072265625,"train/train/tensor_grad_model_layers_24_mlp_gate_proj_weight/mean":-3.864988684654236e-08,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/norm":3.421875,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_52/grad/norm":0.040338220784171645,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/std":3.372705001206065e-05,"train/train/tensor_act_model_layers_47_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/max_abs":0.25,"train/train/layer_model_layers_69/act/mean":0.0003540175301688058,"train/train/tensor_act_model_layers_65/norm":7932.050114516812,"train/train/layer_model_layers_2/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_down_proj/std":0.09191922542990939,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_mlp_up_proj/mean":0.0128021240234375,"train/train/tensor_act_model_layers_81_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/std":0.035888671875,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/max_abs":0.00128173828125,"train/train/tensor_param_model_layers_24_mlp_gate_proj_weight/norm":4.84375,"train/train/tensor_act_model_layers_24_mlp_gate_proj/mean":0.00818634033203125,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/norm":0.0014995302747586334,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/mean":-7.188646122813225e-08,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/norm":0.004784329985301554,"train/train/tensor_act_model_layers_48_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_86/act/max_abs":12.125,"train/train/tensor_act_model_layers_12_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn/mean":-0.00021839141845703125,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/std":6.880657074110274e-05,"train/train/tensor_act_model_layers_93_self_attn_o_proj/max_abs":2.828125,"train/train/tensor_act_model_layers_68_self_attn/norm":1144.5357810048406,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn_o_proj/max_abs":0.8671875,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/mean":0.00016117095947265625,"train/train/tensor_act_model_layers_67_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_58/grad/norm":0.04939627803846162,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/std":7.26048045434519e-05,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/std":0.00015092647715163516,"train/train/tensor_act_model_layers_7/std":1.2890750349764708,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/max_abs":0.2333984375,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/max_abs":8.249282836914062e-05,"train/train/tensor_act_model_layers_65_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_53/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/norm":4.125,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/std":6.64129938308517e-05,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/mean":-1.889653503894806e-06,"train/train/tensor_act_model_layers_2_self_attn_k_proj/norm":6506.840396560156,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/mean":-0.00029754638671875,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/mean":0.00011920928955078125,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/max_abs":0.150390625,"train/train/layer_model_layers_48/act/mean":-0.0032082966395786832,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/mean":0.0001926422119140625,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp_gate_proj/std":0.24414070424436238,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/max_abs":0.00118255615234375,"train/train/tensor_act_model_layers_24_mlp_down_proj/norm":213.7213267493991,"train/train/tensor_act_model_layers_90_self_attn_o_proj/std":0.23974798373178566,"train/train/tensor_act_model_layers_37_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53/std":1.277350322898962,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/mean":0.00023174285888671875,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/mean":0.0001888275146484375,"train/train/tensor_param_model_layers_54_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_87/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp_up_proj/max_abs":2.140625,"train/train/tensor_act_model_layers_72_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/std":0.04736328125,"train/train/tensor_act_model_layers_42_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89/mean":0.0136566162109375,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/max_abs":0.0012664794921875,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/std":6.702478862361754e-05,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/norm":6.125,"train/train/layer__model_layers_75/param/norm":23.59050060405671,"train/train/layer_model_layers_60/act/max_abs":9.875,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/max_abs":0.0019989013671875,"train/train/tensor_param_model_layers_39_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_12_mlp_gate_proj_weight/std":4.3409572818866224e-05,"train/train/tensor_act_model_layers_69_self_attn/norm":793.9336190313563,"train/train/tensor_act_model_layers_60_post_attention_layernorm/max_abs":5.75,"train/train/tensor_act_model_layers_3_input_layernorm/max_abs":4.9375,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/std":0.037109375,"train/train/tensor_act_model_layers_56_self_attn/mean":-0.0003657341003417969,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/max_abs":0.0027008056640625,"train/train/tensor_act_model_layers_11_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp_gate_proj/std":0.22460945318995768,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/max_abs":0.000705718994140625,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_5/act/max_abs":8.125,"train/train/layer__model_layers_69/param/std":0.05668709544687347,"train/train/tensor_act_model_layers_41_mlp_gate_proj/norm":2580.8227292999054,"train/train/tensor_act_model_layers_79_mlp_up_proj/max_abs":2.828125,"train/train/tensor_param_model_layers_26_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/mean":-0.00011754035949707031,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/mean":-2.705305814743042e-05,"train/train/tensor_act_model_layers_62_mlp_down_proj/max_abs":0.73046875,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/mean":-0.00055694580078125,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/mean":1.4513731002807617e-05,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/norm":5.90625,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn/mean":5.5909156799316406e-05,"train/train/tensor_act_model_layers_84_self_attn_q_proj/norm":7221.578903401699,"train/train/tensor_act_model_layers_15_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_21/grad/norm":0.03903199985121993,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_o_proj/norm":501.76970397443307,"train/train/tensor_act_model_layers_51_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/max_abs":0.000537872314453125,"train/train/layer__model_layers_18/param/max_abs":1,"train/train/tensor_act_model_layers_19_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/std":0.0264892578125,"train/train/tensor_act_model_layers_26/std":1.248053639875653,"train/train/tensor_act_model_layers_0_input_layernorm/norm":5792.379028363401,"train/train/tensor_act_model_layers_17_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_35_input_layernorm/norm":5792.602783213852,"train/train/layer_model_layers_14/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_72/grad/max_abs":0.001373291015625,"train/train/tensor_act_model_layers_67_post_attention_layernorm/mean":0.00646209716796875,"train/train/tensor_act_model_layers_67_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/std":5.909652888438949e-05,"train/train/tensor_act_model_layers_87_mlp/max_abs":2.28125,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn_o_proj/norm":1298.0215039662069,"train/train/tensor_act_model_layers_49_mlp_up_proj/std":0.33984381953874515,"train/train/tensor_param_model_layers_11_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/max_abs":0.130859375,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/mean":1.5661120414733887e-05,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/mean":8.96453857421875e-05,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/std":4.522711328350907e-05,"train/train/tensor_act_model_layers_34_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_input_layernorm/norm":5792.607666017328,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/std":1.84071462601423e-05,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38/max_abs":9.4375,"train/train/layer_model_layers_17/act/mean":-0.010105643953595842,"train/train/tensor_param_model_layers_26_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/std":4.0683091810285235e-05,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/max_abs":0.263671875,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/mean":1.7642974853515625e-05,"train/train/tensor_act_model_layers_41_self_attn_v_proj/max_abs":2.578125,"train/train/tensor_act_model_layers_1_self_attn_o_proj/std":0.01617914217245331,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/norm":4.4375,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/max_abs":0.1943359375,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/mean":2.425163984298706e-06,"train/train/tensor_act_model_layers_42_self_attn_o_proj/std":0.07946882225144182,"train/train/tensor_act_model_layers_73_self_attn_v_proj/std":0.39990324889444934,"train/train/tensor_param_model_layers_86_mlp_gate_proj_weight/max_abs":0.2080078125,"train/train/layer_model_layers_53/act/std":0.6420608034999513,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/norm":7.0625,"train/train/tensor_act_model_layers_69_mlp_gate_proj/norm":3637.6538103620405,"train/train/tensor_param_model_layers_38_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/std":4.2929913002150616e-05,"train/train/tensor_act_model_layers_28_self_attn/norm":559.808692063593,"train/train/tensor_param_model_layers_29_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/max_abs":0.00077056884765625,"train/train/tensor_act_model_layers_9_self_attn_o_proj/std":0.0692176483655624,"train/train/tensor_act_model_layers_32_self_attn_q_proj/max_abs":5.40625,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/norm":5.625,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_13_mlp_gate_proj/std":0.23803749176508554,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/mean":3.2689422369003296e-07,"train/train/tensor_act_/std":0,"train/train/tensor_act_model_layers_7_mlp_up_proj/max_abs":2.125,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/max_abs":0.00021457672119140625,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/mean":0.0001659393310546875,"train/train/tensor_act_model_layers_7_mlp_gate_proj/norm":1798.0342640308359,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/mean":-4.736357368528843e-07,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/max_abs":0.2060546875,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_68_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/norm":0.026827225138481108,"train/train/tensor_act_model_layers_13_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/norm":0.0012505517892823779,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/max_abs":0.00159454345703125,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/mean":6.29425048828125e-05,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/mean":0.0001506805419921875,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/max_abs":0.000667572021484375,"train/train/tensor_act_model_layers_21_mlp_up_proj/mean":-0.002918243408203125,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/max_abs":0.0002765655517578125,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/norm":3.21875,"train/train/layer__model_layers_52/param/std":0.052984529982163714,"train/train/tensor_act_model_layers_6_self_attn_v_proj/norm":1601.0538692539158,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/std":0.0233154296875,"train/train/tensor_act_model_layers_84_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_o_proj/norm":1375.3179945246777,"train/train/tensor_act_model_layers_35_self_attn_k_proj/norm":4713.833590141195,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/std":0.049072265625,"train/train/layer_model_layers_42/grad/std":5.0785638501982266e-05,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/std":2.3343843334522875e-05,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/norm":0.023663149821431462,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/std":0.03271484375,"train/train/tensor_act_model_layers_33_self_attn_k_proj/std":0.8623079420559683,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp_gate_proj/mean":-0.0099334716796875,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/norm":0.0008687297782242541,"train/train/tensor_act_model_layers_70_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/mean":2.7526402845978737e-07,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/mean":-1.5027821063995361e-05,"train/train/tensor_param_model_layers_16_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_83_self_attn/max_abs":2.28125,"train/train/tensor_act_model_layers_93_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_v_proj/std":0.3105469151112993,"train/train/layer__model_layers_62/param/std":0.05519661466371064,"train/train/tensor_param_model_layers_40_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_69_self_attn_o_proj/std":0.1369670630949227,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/norm":0.005636878386078911,"train/train/layer_model_layers_39/act/mean":-0.009454454694475447,"train/train/layer_model_layers_18/act/mean":-0.010390630790165492,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_mlp_gate_proj_weight/norm":5.96875,"train/train/tensor_act_model_layers_24_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/norm":0.017130666748884506,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/std":1.4092740672166755e-05,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp/norm":245.78391475750408,"train/train/tensor_act_model_layers_47_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_post_attention_layernorm/std":1.0000000348081806,"train/train/tensor_act_model_layers_12_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_mlp_gate_proj_weight/norm":5.5625,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/mean":0.0002498626708984375,"train/train/tensor_grad_model_layers_41_mlp_gate_proj_weight/max_abs":0.000518798828125,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/norm":0.05332936080900061,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/std":0.03076171875,"train/train/tensor_act_model_layers_1_mlp_up_proj/norm":3028.8124233362937,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_post_attention_layernorm/max_abs":5.03125,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/mean":-0.0002994537353515625,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/std":6.724341846126035e-05,"train/train/tensor_act_model_layers_33_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/max_abs":0.267578125,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/std":0.047607421875,"train/train/tensor_act_model_layers_69_mlp/norm":781.7845838840727,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_68/act/mean":-0.009326117379324777,"train/train/tensor_act_model_layers_10_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/mean":3.852182999253273e-07,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_post_attention_layernorm/max_abs":5.75,"train/train/tensor_grad_model_layers_87_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_input_layernorm/std":1.0000015107423181,"train/train/tensor_act_model_layers_74_input_layernorm/max_abs":5.78125,"train/train/tensor_grad_model_layers_46_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp_gate_proj/mean":0.000530242919921875,"train/train/tensor_act_model_layers_56_mlp/std":0.08483915448218864,"train/train/tensor_act_model_layers_64_mlp/norm":619.0999122773325,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/std":0.00013480081636936262,"train/train/tensor_grad_model_layers_32_mlp_gate_proj_weight/norm":0.015861538753513166,"train/train/tensor_grad_model_layers_72_mlp_gate_proj_weight/std":7.28499393868503e-05,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/mean":-0.000286102294921875,"train/train/tensor_act_model_layers_54_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_up_proj/max_abs":1.6328125,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_gate_proj/mean":-0.002079010009765625,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/std":9.613045847066181e-05,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/norm":0.004145289540020552,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/std":0.052001953125,"train/train/tensor_act_model_layers_28_mlp/norm":224.18272518682747,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/mean":8.149072527885437e-10,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/mean":7.258495315909386e-08,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/mean":0.00026702880859375,"train/train/tensor_act_model_layers_32_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_mlp_down_proj/norm":259.868483762349,"train/train/tensor_act_model_layers_6_input_layernorm/norm":5792.604248053774,"train/train/tensor_param_model_layers_45_mlp_gate_proj_weight/mean":-2.9802322387695312e-05,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/max_abs":0.00102996826171875,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/std":1.4342735094759271e-05,"train/train/tensor_param_model_layers_40_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/std":0.0260009765625,"train/train/tensor_act_model_layers_73_self_attn/mean":0.0014586448669433594,"train/train/tensor_act_model_layers_9_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/norm":7.71875,"train/train/tensor_act_model_layers_81_self_attn_v_proj/std":0.4648439561744241,"train/train/tensor_act_model_layers_79_mlp_gate_proj/std":0.4941406445112996,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/norm":0.005560504562115108,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_gate_proj_weight/norm":4.625,"train/train/tensor_act_model_layers_11_self_attn_o_proj/norm":303.4659733248781,"train/train/tensor_param_model_layers_81_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_61_mlp_up_proj/std":0.40429712959741676,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/mean":3.5390257835388184e-08,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/max_abs":0.0004119873046875,"train/train/tensor_act_model_layers_17/norm":7390.797461184207,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/std":0.029052734375,"train/train/tensor_act_model_layers_39_mlp/mean":0.000980377197265625,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/mean":2.812594175338745e-07,"train/train/tensor_act_model_layers_64_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_k_proj/mean":0.0015468597412109375,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/mean":-0.000362396240234375,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/std":5.481385478013937e-05,"train/train/layer_model_layers_85/grad/max_abs":0.00127410888671875,"train/train/tensor_act_model_layers_17_self_attn/norm":335.9912741121645,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/std":0.03271484375,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/norm":5.875,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/norm":0.019945612957817223,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/std":5.4490334550370895e-05,"train/train/tensor_act_model_layers_32_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_input_layernorm/max_abs":6.125,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/norm":0.000651224801015678,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/norm":3.25,"train/train/layer_model_layers_73/act/mean":-0.004743916647774833,"train/train/tensor_act_model_layers_32_mlp_gate_proj/mean":-0.0087738037109375,"train/train/tensor_act_model_layers_69_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp_gate_proj/std":0.5976563724719496,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_v_proj/max_abs":1.4453125,"train/train/tensor_act_model_layers_30_post_attention_layernorm/std":1.0000009819627576,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_46_mlp_gate_proj_weight/mean":-7.62939453125e-05,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/max_abs":0.11376953125,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/std":0.041748046875,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93/mean":0.0114593505859375,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/norm":0.010040468718603354,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/norm":5.53125,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/max_abs":0.166015625,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/mean":-8.068978786468506e-06,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/norm":0.02153338996650445,"train/train/layer_model_layers_1/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/max_abs":0.0010833740234375,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/std":6.5096287474392e-05,"train/train/tensor_act_model_layers_16_mlp_down_proj/norm":206.55383678351737,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/std":0.0296630859375,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/std":2.3280772485426318e-05,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_91/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp/frac_near_user_limit":0,"train/train/layer__model_layers_10/param/std":0.04903007584050826,"train/train/tensor_act_model_layers_31_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp_down_proj/norm":1111.8647286990276,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/norm":0.0013606121973297072,"train/train/tensor_act_model_layers_78_mlp_up_proj/norm":4028.7978329343428,"train/train/layer_model_layers_34/act/max_abs":9.3125,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/norm":0.004626660103816733,"train/train/layer__model_layers_60/param/max_abs":1,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/std":0.00011003696321103817,"train/train/tensor_act_model_layers_87_self_attn_q_proj/mean":0.07373046875,"train/train/tensor_act_model_layers_39_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/norm":5.75,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_gate_proj_weight/norm":4.34375,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/max_abs":0.00032806396484375,"train/train/tensor_act_model_layers_19_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/mean":0.06219482421875,"train/train/tensor_grad_model_layers_37_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp/max_abs":0.47265625,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/max_abs":0.11572265625,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_29/param/norm":20.569818463968634,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp_up_proj/std":0.4921875245987417,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/mean":1.9581057131290436e-07,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/std":0.00010106458817207751,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/mean":0.000118255615234375,"train/train/tensor_act_model_layers_5_post_attention_layernorm/norm":5792.605590826315,"train/train/tensor_act_model_layers_39_mlp_down_proj/max_abs":0.46484375,"train/train/tensor_act_model_layers_40_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/mean":-0.0050811767578125,"train/train/tensor_act_model_layers_55_self_attn_o_proj/std":0.0639649144571616,"train/train/tensor_act_model_layers_81_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/std":0.0361328125,"train/train/tensor_act_model_layers_84_self_attn_v_proj/mean":-0.00040628015995025635,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/mean":3.5762786865234375e-05,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/mean":5.459785461425781e-05,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/mean":0.0003032684326171875,"train/train/tensor_act_model_layers_45_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_74/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_post_attention_layernorm/std":1.0000003217718978,"train/train/tensor_param_model_layers_77_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_input_layernorm/norm":5792.611328135567,"train/train/tensor_act_model_layers_87_self_attn_v_proj/norm":2707.1019127721283,"train/train/layer__model_layers_82/param/mean":0.0013261101733131825,"train/train/tensor_param_model_layers_35_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_33_mlp_gate_proj_weight/norm":0.01624292182822012,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/std":1.5864039120419597e-05,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/std":0.0302734375,"train/train/tensor_act_model_layers_40_self_attn_v_proj/max_abs":2.546875,"train/train/tensor_act_model_layers_90_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/mean":-0.0003719329833984375,"train/train/tensor_param_model_layers_70_mlp_gate_proj_weight/norm":6.96875,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/std":8.91383599745506e-05,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/mean":-2.6635825634002686e-07,"train/train/layer_model_layers_76/act/mean":0.003493411200387137,"train/train/tensor_act_model_layers_1_self_attn_o_proj/max_abs":0.314453125,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/norm":0.003711933704977595,"train/train/tensor_act_model_layers_83_mlp_up_proj/std":0.5351564915184012,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/mean":0.0002765655517578125,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_o_proj/mean":0.0014586448669433594,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/std":4.3117818050628614e-05,"train/train/layer_model_layers_43/act/mean":-0.003623776137828827,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/std":6.662888924613025e-05,"train/train/layer_model_layers_28/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/norm":0.024912215127574685,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_post_attention_layernorm/std":1.000000707106415,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/std":0.00013284975053226252,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/std":0.00014523096671792272,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/std":0.028076171875,"train/train/tensor_act_model_layers_65_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/norm":6.1875,"train/train/tensor_param_model_layers_54_mlp_gate_proj_weight/mean":0.0002803802490234375,"train/train/tensor_act_model_layers_52_mlp/mean":0.0007600784301757812,"train/train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/mean":-8.344650268554688e-05,"train/train/tensor_act_model_layers_14_self_attn/max_abs":0.65625,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/max_abs":0.0013885498046875,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_gate_proj_weight/norm":6.1875,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/norm":0.013074921483374852,"train/train/layer__model_layers_9/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn_q_proj/max_abs":7.09375,"train/train/tensor_act_model_layers_9_post_attention_layernorm/norm":5792.612670900561,"train/train/tensor_act_model_layers_22_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/max_abs":0.0002288818359375,"train/train/tensor_act_model_layers_67/std":1.4140649644317478,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/mean":-6.575137376785278e-07,"train/train/layer_model_layers_67/grad/max_abs":0.0013275146484375,"train/train/tensor_act_model_layers_11_post_attention_layernorm/mean":-0.0296630859375,"train/train/layer__model_layers_47/param/mean":0.0015765947410357351,"train/train/tensor_act_model_layers_48_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/mean":-0.0004138946533203125,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/std":0.0478515625,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_6/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn/mean":-0.0010061264038085938,"train/train/layer_model_layers_65/grad/max_abs":0.0017242431640625,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/std":6.561582935865426e-05,"train/train/layer__model_layers_14/param/std":0.049237485699975186,"train/train/tensor_act_model_layers_14/norm":7413.676951558679,"train/train/tensor_act_model_layers_75_post_attention_layernorm/max_abs":5.65625,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/norm":6.75,"train/train/layer_model_layers_83/grad/std":8.159071648548239e-05,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/norm":0.015036621143299567,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/norm":0.029391263862573822,"train/train/layer_model_layers_7/grad/mean":-9.975991703645441e-08,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/norm":0.001376439829108858,"train/train/tensor_act_model_layers_30_mlp_down_proj/norm":242.92867394544308,"train/train/tensor_param_model_layers_7_mlp_gate_proj_weight/norm":4.40625,"train/train/tensor_act_model_layers_35_self_attn_q_proj/max_abs":4.84375,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_input_layernorm/std":1.0000013834787276,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/norm":8.6875,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_up_proj/norm":1894.179177983833,"train/train/tensor_act_model_layers_32_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/norm":3.4375,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/std":3.331928131218764e-05,"train/train/tensor_act_model_layers_64_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_down_proj/max_abs":0.95703125,"train/train/tensor_act_model_layers_42_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/mean":1.9431114196777344e-05,"train/train/layer_model_layers_59/act/mean":0.010133811405726842,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/mean":-2.849847078323364e-07,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/std":0.0341796875,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/max_abs":0.000957489013671875,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/max_abs":0.00179290771484375,"train/train/layer_model_layers_24/grad/norm":0.029321271112860953,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/max_abs":0.0004367828369140625,"train/train/tensor_param_model_layers_8_mlp_gate_proj_weight/max_abs":0.1064453125,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/norm":0.020634938359835167,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/norm":4.8125,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/norm":0.017800409162029903,"train/train/tensor_act_model_layers_7_input_layernorm/norm":5792.611694338915,"train/train/tensor_param_model_layers_35_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer__model_layers_9/param/std":0.048183210571053396,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/norm":0.0028865741363147132,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_o_proj/max_abs":1.59375,"train/train/tensor_act_model_layers_2_self_attn_v_proj/norm":1388.4196334706994,"train/train/tensor_param_model_layers_46_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_92/grad/std":8.68485766307368e-05,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/std":8.506001013256655e-05,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_22/grad/std":4.0527007599953585e-05,"train/train/tensor_param_model_layers_31_mlp_gate_proj_weight/max_abs":0.130859375,"train/train/tensor_param_model_layers_45_mlp_gate_proj_weight/max_abs":0.130859375,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/std":5.093240481344821e-05,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/max_abs":0.208984375,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/norm":6.3125,"train/train/tensor_act_model_layers_55/max_abs":10.0625,"train/train/tensor_act_model_layers_12_mlp_down_proj/max_abs":0.69921875,"train/train/layer_model_layers_36/act/mean":-0.006593908582414899,"train/train/tensor_act_model_layers_20_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/mean":-0.00023746490478515625,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/std":8.717064687353886e-05,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/norm":0.014088812688546405,"train/train/tensor_act_model_layers_84_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/std":0.0263671875,"train/train/tensor_act_model_layers_24_self_attn_o_proj/norm":162.69204488932456,"train/train/tensor_act_model_layers_21_mlp_down_proj/std":0.04119887351964911,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/max_abs":0.306640625,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/max_abs":0.0003986358642578125,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/mean":0.000240325927734375,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_34/grad/mean":1.7682415856028125e-08,"train/train/tensor_param_model_layers_67_input_layernorm_weight/std":0,"train/train/layer_model_layers_2/act/max_abs":8,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/norm":4.375,"train/train/tensor_param_model_layers_91_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/grad/max_abs":0.0019989013671875,"train/train/tensor_act_model_layers_50_self_attn_o_proj/std":0.0933866118792555,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/std":0.00013242946708311626,"train/train/tensor_act_model_layers_30_mlp_gate_proj/mean":0.00018393993377685547,"train/train/tensor_act_model_layers_55/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_mlp_gate_proj_weight/norm":0.014715992602449119,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/max_abs":0.00106048583984375,"train/train/tensor_act_model_layers_38_self_attn_k_proj/max_abs":3.96875,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/mean":0.00020694732666015625,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_55/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/norm":5.34375,"train/train/tensor_act_model_layers_26_mlp/mean":0.001064300537109375,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/norm":0.0007696382171697861,"train/train/tensor_act_model_layers_39_self_attn_q_proj/max_abs":6.4375,"train/train/tensor_act_model_layers_86_self_attn_k_proj/max_abs":5.5625,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/norm":7.15625,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/norm":0.01595310592463415,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/max_abs":0.00119781494140625,"train/train/layer__model_layers_2/param/max_abs":1,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/norm":0.01395457764055761,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/mean":-1.043081283569336e-05,"train/train/tensor_act_model_layers_50_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/mean":-1.1714291758835316e-08,"train/train/tensor_act_model_layers_52/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_63/grad/norm":0.05231636103878262,"train/train/tensor_act_model_layers_78_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/norm":0.0014087445101984301,"train/train/tensor_act_model_layers_11_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_o_proj/std":0.2243809738919304,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_54/grad/mean":7.26642281793395e-08,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/max_abs":0.00012683868408203125,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/mean":-4.866160452365875e-07,"train/train/tensor_act_model_layers_91_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/max_abs":0.2236328125,"train/train/tensor_act_model_layers_6_self_attn_q_proj/mean":-0.03179931640625,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/std":0.039306640625,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/mean":-2.0638108253479004e-06,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/std":0.045166015625,"train/train/tensor_act_model_layers_23_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_50/act/mean":-0.0047574639320373535,"train/train/tensor_act_model_layers_59/mean":-0.00010585784912109375,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/mean":5.543231964111328e-06,"train/train/tensor_param_model_layers_91_mlp_gate_proj_weight/norm":9.8125,"train/train/tensor_act_model_layers_61_mlp/max_abs":0.7890625,"train/train/tensor_act_model_layers_9_self_attn_k_proj/mean":-0.0218505859375,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_73_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/max_abs":0.1328125,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/max_abs":0.00069427490234375,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_post_attention_layernorm/norm":5792.614624023853,"train/train/tensor_param_model_layers_51_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16/std":1.279302889322019,"train/train/tensor_act_model_layers_75_mlp_gate_proj/max_abs":2.734375,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/std":5.9852520212582154e-05,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/max_abs":0.189453125,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/mean":-8.344650268554688e-05,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/norm":6.3125,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_51/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/std":4.1075089164053435e-05,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_29/act/norm":13732.580011879425,"train/train/tensor_act_model_layers_60_self_attn_o_proj/std":0.08874606221098551,"train/train/tensor_act_model_layers_58_mlp/frac_near_user_limit":0,"train/train/layer__model_layers_43/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn_q_proj/max_abs":4.875,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_o_proj/max_abs":1.40625,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/max_abs":0.0013580322265625,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_k_proj/max_abs":4.875,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_89_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/norm":0.029425909080820198,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/std":7.346052270869022e-05,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_79_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_25/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/max_abs":0.00038909912109375,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/max_abs":0.287109375,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/max_abs":0.000347137451171875,"train/train/tensor_act_model_layers_64/norm":7796.89289668641,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/max_abs":0.216796875,"train/train/tensor_act_model_layers_81_self_attn_o_proj/mean":-0.00014060735702514648,"train/train/tensor_act_model_layers_76_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_post_attention_layernorm/std":1.000001726757874,"train/train/tensor_act_model_layers_42_self_attn_o_proj/norm":460.7705034637365,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_67/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_o_proj/mean":0.00234222412109375,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/norm":6,"train/train/tensor_act_model_layers_49_self_attn_k_proj/norm":4436.281855427093,"train/train/tensor_param_model_layers_42_mlp_gate_proj_weight/mean":-5.125999450683594e-05,"train/train/tensor_act_model_layers_66_mlp_up_proj/mean":0.0157928466796875,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_gate_proj_weight/mean":-6.0558319091796875e-05,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/norm":6.5625,"train/train/tensor_param_model_layers_25_mlp_gate_proj_weight/mean":2.2351741790771484e-06,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31/mean":-0.018707275390625,"train/train/tensor_act_model_layers_76_self_attn_o_proj/std":0.17260803759536225,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/mean":0.00038909912109375,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/std":0.03857421875,"train/train/tensor_act_model_layers_91_post_attention_layernorm/std":1.0000011213117512,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/mean":7.953494787216187e-06,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/mean":9.834766387939453e-07,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/mean":-4.190951585769653e-09,"train/train/layer_model_layers_93/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_mlp_gate_proj_weight/max_abs":0.0008392333984375,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/norm":0.025669836849166742,"train/train/tensor_act_model_layers_66_self_attn_k_proj/std":0.9091855913672123,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_k_proj/norm":5406.69215760537,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_77/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_38/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/std":7.598401042274039e-05,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/std":0.04248046875,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/mean":1.328007783740759e-07,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/max_abs":0.00022029876708984375,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/std":0.032470703125,"train/train/tensor_grad_model_layers_55_mlp_gate_proj_weight/max_abs":0.000614166259765625,"train/train/tensor_act_model_layers_54/norm":7469.085598028393,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/mean":-1.9907020032405853e-08,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/mean":-3.259629011154175e-08,"train/train/tensor_act_model_layers_27_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/mean":-3.007054328918457e-05,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/mean":1.4983117580413818e-05,"train/train/tensor_act_model_layers_41_post_attention_layernorm/max_abs":6.625,"train/train/tensor_act_model_layers_37_mlp/mean":0.000457763671875,"train/train/tensor_param_model_layers_19_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/mean":1.917942427098751e-08,"train/train/layer_model_layers_34/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/norm":0.014595125460286388,"train/train/tensor_act_model_layers_65_mlp/norm":678.5794164842906,"train/train/tensor_act_model_layers_67_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/mean":-5.269050598144531e-05,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/norm":0.012699566641870566,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/max_abs":0.23046875,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/std":2.813599518741507e-05,"train/train/tensor_act_model_layers_31_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_o_proj/std":0.1399052024705307,"train/train/tensor_act_model_layers_17_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_gate_proj/std":0.5771511901362811,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_11_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/mean":-8.777715265750885e-08,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/norm":0.0009740241409994919,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/norm":7.4375,"train/train/layer_model_layers_54/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp_down_proj/std":0.04425063116774016,"train/train/tensor_act_model_layers_51_self_attn_k_proj/norm":4446.680150466528,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/norm":5.65625,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/mean":3.189779818058014e-07,"train/train/tensor_act_model_layers_11/std":1.2871144270038009,"train/train/tensor_param_model_layers_87_mlp_gate_proj_weight/mean":-0.00021839141845703125,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_mlp_gate_proj_weight/max_abs":0.000640869140625,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/max_abs":0.0002918243408203125,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/max_abs":0.0002193450927734375,"train/train/tensor_act_model_layers_19_mlp_gate_proj/norm":1994.186107749807,"train/train/tensor_act_model/std":1.000000642728785,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/norm":3.90625,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/std":0.032958984375,"train/train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_up_proj/norm":3652.025225391881,"train/train/tensor_act_model_layers_46_mlp_down_proj/norm":369.4766324211434,"train/train/tensor_act_model_layers_59_mlp_gate_proj/mean":-0.0002815723419189453,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/std":0.0245361328125,"train/train/tensor_act_model_layers_46_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn_o_proj/std":0.2822284554051737,"train/train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_q_proj/max_abs":6.375,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/max_abs":0.00046539306640625,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/max_abs":0.166015625,"train/train/tensor_act_model_layers_77_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_59/grad/max_abs":0.0013580322265625,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/max_abs":0.000553131103515625,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/max_abs":0.1591796875,"train/train/tensor_act_model_layers_68_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/mean":2.3469328880310059e-07,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/norm":3.890625,"train/train/layer__model_layers_81/param/max_abs":1,"train/train/layer_model_layers_26/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_78/grad/std":8.048566914636184e-05,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_q_proj/std":0.960937569718667,"train/train/tensor_act_model_layers_71_self_attn_o_proj/mean":-0.0011615753173828125,"train/train/tensor_act_model_layers_57_mlp_gate_proj/norm":3136.397691801071,"train/train/tensor_act_model_layers_24_input_layernorm/norm":5792.606811524779,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/norm":4.03125,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_55/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_56/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/mean":0.0002269744873046875,"train/train/tensor_param_model_layers_16_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/mean":2.8742942959070206e-07,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/std":8.369446785791249e-05,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/norm":7.875,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/norm":0.020555350292601443,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/max_abs":0.000499725341796875,"train/train/tensor_act_model_layers_21_input_layernorm/max_abs":6.3125,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_60/grad/norm":0.05013185569091776,"train/train/tensor_act_model_layers_88_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/std":2.3283940231687098e-05,"train/train/tensor_grad_model_layers_42_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/mean":2.942979335784912e-07,"train/train/layer_model_layers_57/grad/norm":0.055848330996342406,"train/train/tensor_act_model_layers_48_mlp_down_proj/mean":0.0012569427490234375,"train/train/tensor_act_model_layers_62_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/norm":0.003217958870437545,"train/train/tensor_act_model_layers_74_mlp/max_abs":1.15625,"train/train/tensor_param_model_layers_88_mlp_gate_proj_weight/std":0.048095703125,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/std":0.043701171875,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/std":6.677021675299011e-05,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/mean":-2.2351741790771484e-06,"train/train/tensor_act_model_layers_74_mlp/std":0.1494140780220421,"train/train/tensor_act_model_layers_90_post_attention_layernorm/std":1.0000010869401652,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/std":2.837829951397764e-05,"train/train/tensor_param_model_layers_32_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_38_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/norm":0.0034094773126839765,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/max_abs":0.140625,"train/train/tensor_act_model_layers_3_self_attn_v_proj/max_abs":1.6171875,"train/train/tensor_param_model_layers_93_mlp_gate_proj_weight/std":0.058837890625,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn_k_proj/norm":4604.618720512917,"train/train/tensor_act_model_layers_61_input_layernorm/mean":-0.0001971423625946045,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_mlp_up_proj/max_abs":1.890625,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/mean":6.467103958129883e-06,"train/train/tensor_act_model_layers_92_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_75/grad/max_abs":0.0013885498046875,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/max_abs":0.0004520416259765625,"train/train/tensor_act_model_layers_9/max_abs":8.6875,"train/train/tensor_act_model_layers_42_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/std":0.04541015625,"train/train/tensor_act_model_layers_16_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/norm":7.59375,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/std":0.04345703125,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/mean":-1.1670636013150215e-07,"train/train/tensor_act_model_layers_92_mlp_down_proj/max_abs":5.84375,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/mean":-0.000278472900390625,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_87/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/std":3.650414839391795e-05,"train/train/tensor_act_model_layers_46_mlp_gate_proj/mean":-0.00560760498046875,"train/train/tensor_act_model_layers_59_mlp_down_proj/norm":579.4086061424512,"train/train/layer_model_layers_14/act/mean":-0.008861788681575231,"train/train/tensor_act_model_layers_87/frac_near_user_limit":0,"train/train/layer_model_layers_68/act/norm":15259.617211459194,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/mean":0.00010728836059570312,"train/train/tensor_act_model_layers_8_self_attn_o_proj/norm":245.47060492655476,"train/train/tensor_act_model_layers_88_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_mlp/std":0.047120097590244485,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/mean":9.284121915698051e-08,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/norm":0.0030970361619302895,"train/train/tensor_act_model_layers_53_self_attn_o_proj/std":0.10095430427195932,"train/train/tensor_act_model_layers_70_mlp/max_abs":1.09375,"train/train/tensor_act_model_layers_7_mlp/mean":2.2761523723602295e-05,"train/train/tensor_param_model_layers_2_mlp_gate_proj_weight/std":0.0235595703125,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88/std":2.152356396532377,"train/train/tensor_act_model_layers_18_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/mean":0.0002593994140625,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/norm":0.014796997780778882,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/std":3.006106206938213e-05,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/norm":0.03774835678970512,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/max_abs":0.294921875,"train/train/tensor_act_model_layers_34_mlp_up_proj/mean":0.0088653564453125,"train/train/tensor_act_model_layers_48_self_attn_k_proj/norm":4320.060800934029,"train/train/tensor_act_model_layers_12_input_layernorm/mean":-0.029205322265625,"train/train/tensor_act_model_layers_12_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/norm":0.015398923166590929,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/std":3.109460419996191e-05,"train/train/tensor_param_model_layers_79_mlp_gate_proj_weight/norm":7.59375,"train/train/tensor_act_model_layers_36_self_attn_k_proj/std":0.878906658490404,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/mean":5.461275577545166e-06,"train/train/layer__model_layers_84/param/norm":24.849291834728408,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61/mean":-0.001199483871459961,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_61_mlp_gate_proj_weight/max_abs":0.000652313232421875,"train/train/tensor_act_model_layers_65_input_layernorm/norm":5792.6180419937045,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/std":2.0944860783997115e-05,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/std":4.00162520182867e-05,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_mlp_gate_proj_weight/std":4.01986448951138e-05,"train/train/tensor_act_model_layers_52_self_attn/max_abs":0.59765625,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/norm":0.005923673379337748,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_70/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_62/act/mean":0.005456204925264631,"train/train/tensor_act_model_layers_55_mlp_up_proj/max_abs":2.25,"train/train/tensor_act_model_layers_47_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_mlp/max_abs":0.5390625,"train/train/tensor_act_model_layers_10_mlp/mean":-0.0002684593200683594,"train/train/layer__model_layers_10/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_54_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/std":0.042236328125,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/max_abs":0.0003376007080078125,"train/train/tensor_act_model_layers_45_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_73/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp_up_proj/std":0.3261721508944367,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/max_abs":0.00022029876708984375,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/norm":0.003859915711868971,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/max_abs":0.1474609375,"train/train/layer_model_layers_8/act/std":0.6447277669681104,"train/train/layer_model_layers_29/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_k_proj/max_abs":5.40625,"train/train/layer__model_layers_65/param/std":0.059226115214167895,"train/train/tensor_act_model_layers_92_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/max_abs":0.16796875,"train/train/tensor_act_model_layers_85_mlp_up_proj/max_abs":3.328125,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/norm":0.006684986767498978,"train/train/tensor_act_model_layers_32_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_13/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_gate_proj/max_abs":2.484375,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_o_proj/norm":409.3614131422575,"train/train/tensor_param_model_layers_2_mlp_gate_proj_weight/norm":4.25,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/std":3.9886484190454946e-05,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/mean":1,"train/train/layer__model_layers_71/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn_o_proj/norm":469.32216258012346,"train/train/tensor_act_model_layers_15_self_attn/mean":-0.0006361007690429688,"train/train/tensor_grad_model_layers_88_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/mean":-1.4994293451309204e-07,"train/train/layer__model_layers_82/param/max_abs":1,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_k_proj/norm":4686.183944009987,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/mean":-1.4354554878082126e-08,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn_q_proj/std":1.0703126923010995,"train/train/tensor_param_model_layers_34_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/std":6.866616885431456e-05,"train/train/tensor_act_model_layers_7_self_attn_q_proj/norm":5546.234069082709,"train/train/tensor_act_model_layers_59_mlp_up_proj/std":0.3969735665211206,"train/train/tensor_act_model_layers_15_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/norm":3.65625,"train/train/tensor_act_model_layers_67_input_layernorm/norm":5792.608886723906,"train/train/layer_model_layers_41/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_24/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_mlp_gate_proj_weight/mean":0.00010824203491210938,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/mean":0.000133514404296875,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/std":5.8768229315135454e-05,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/std":0.0260009765625,"train/train/tensor_grad_model_layers_89_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_mlp_gate_proj_weight/std":0.024658203125,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_90_mlp_gate_proj_weight/mean":0.00020885467529296875,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/mean":-1.4632940292358398e-05,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp_down_proj/max_abs":1.703125,"train/train/tensor_act_model_layers_33_self_attn_v_proj/norm":1845.4772615747822,"train/train/layer__model_layers_51/param/mean":0.0015746784656542512,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_20/act/std":0.6116814785306378,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/max_abs":0.1171875,"train/train/tensor_act_model_layers_44_mlp_up_proj/max_abs":2.078125,"train/train/tensor_grad_model_layers_15_mlp_gate_proj_weight/norm":0.014853034127661903,"train/train/layer_model_layers_55/act/max_abs":10.0625,"train/train/tensor_act_model_layers_42/max_abs":9.3125,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/norm":0.013323186133728912,"train/train/tensor_act_model_layers_65_mlp/max_abs":1.0546875,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/std":0.045166015625,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/std":5.638485408763294e-05,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/max_abs":0.0004367828369140625,"train/train/tensor_act_model_layers_7_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/max_abs":0.00014495849609375,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/std":0.0264892578125,"train/train/tensor_act_model_layers_62_mlp/std":0.10107423570614328,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/std":6.837156015868037e-05,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/norm":6.6875,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/norm":0.013371993457815982,"train/train/layer_model_layers_16/grad/norm":0.046247945663554335,"train/train/tensor_act_model_layers_78_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_q_proj/max_abs":7.375,"train/train/tensor_param_model_layers_26_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_78/grad/mean":9.314097498954738e-08,"train/train/tensor_act_model_layers_84/norm":11148.489455496647,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_16_self_attn_v_proj/std":0.3017595381046883,"train/train/tensor_act_model_layers_51_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_mlp_gate_proj_weight/std":8.535876693470305e-05,"train/train/tensor_act_model_layers_17_self_attn_o_proj/norm":335.9912741121645,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/std":0.037353515625,"train/train/tensor_act_model_layers_51_mlp_up_proj/norm":2902.671774597392,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/mean":5.14984130859375e-05,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_input_layernorm/mean":-3.337860107421875e-05,"train/train/tensor_act_model_layers_44_input_layernorm/std":1.0000002598389646,"train/train/tensor_act_model_layers_49_input_layernorm/max_abs":6.1875,"train/train/tensor_act_model_layers_52_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_16/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_50/act/std":0.635897829242168,"train/train/tensor_act_model_layers_6/mean":-0.03192138671875,"train/train/tensor_act_model_layers_62/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/norm":5.125,"train/train/tensor_act_model_layers_28_mlp/std":0.03875747248795357,"train/train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_q_proj/mean":-0.018157958984375,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/mean":-7.867813110351562e-05,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn_k_proj/max_abs":4.9375,"train/train/tensor_act_model_layers_33_self_attn_q_proj/mean":-0.01409912109375,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/mean":2.8405338525772095e-07,"train/train/tensor_act_model_layers_20/mean":-0.02960205078125,"train/train/tensor_act_model_layers_65_mlp_gate_proj/norm":3475.9398588911245,"train/train/layer__model_layers_62/param/frac_near_user_limit":0,"train/train/layer__model_layers_79/param/std":0.0594816306478105,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/max_abs":0.0002956390380859375,"train/train/tensor_param_model_layers_66_input_layernorm_weight/max_abs":1,"train/train/layer__model_layers_32/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/norm":0.00406646728515625,"train/train/tensor_act_model_layers_51_mlp/max_abs":0.5703125,"train/train/tensor_act_model_layers_81_input_layernorm/mean":0.006214141845703125,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/std":0.00015812948496960017,"train/train/tensor_param_model_layers_81_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_1/param/norm":18.70070342510543,"train/train/tensor_act_model_layers_48_mlp/std":0.06555210286920718,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/mean":0.00010585784912109375,"train/train/tensor_act_model_layers_12_mlp_up_proj/max_abs":1.890625,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_post_attention_layernorm/max_abs":6.28125,"train/train/layer__model_layers_29/param/mean":0.001611885154117102,"train/train/tensor_act_model_layers_19_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/max_abs":0.0002498626708984375,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/std":5.045625905982345e-05,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/mean":6.492249667644501e-06,"train/train/tensor_act_model_layers_28_post_attention_layernorm/std":1.0000011818476489,"train/train/layer__model_layers_38/param/norm":21.305740313645522,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/std":0.0224609375,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/max_abs":0.1376953125,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/max_abs":0.2431640625,"train/train/layer_model_layers_32/grad/std":4.792074478745239e-05,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_input_layernorm/mean":-0.0144805908203125,"train/train/tensor_act_model_layers_65_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_mlp_gate_proj_weight/norm":7.28125,"train/train/tensor_act_model_layers_45_self_attn_k_proj/std":0.7373066807397842,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/max_abs":0.216796875,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_input_layernorm/norm":5792.608520512055,"train/train/tensor_grad_model_layers_27_mlp_gate_proj_weight/max_abs":0.000362396240234375,"train/train/layer_model_layers_53/grad/max_abs":0.00115203857421875,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/max_abs":0.00022029876708984375,"train/train/tensor_act_model_layers_92_self_attn_o_proj/norm":1470.4452309839792,"train/train/tensor_act_model_layers_11_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/mean":-1.3515818864107132e-07,"train/train/tensor_param_model_layers_29_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_61_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn/norm":221.1266481226035,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/norm":5.25,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/max_abs":0.000518798828125,"train/train/tensor_act_model_layers_4_self_attn_v_proj/std":0.3574219898936342,"train/train/tensor_act_model_layers_16_self_attn_v_proj/max_abs":1.9375,"train/train/layer__model_layers_64/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_input_layernorm/norm":5792.611694339821,"train/train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn_v_proj/max_abs":4.375,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/std":4.799947529194163e-05,"train/train/tensor_act_model_layers_63_mlp_up_proj/mean":0.0132598876953125,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/max_abs":0.1708984375,"train/train/tensor_act_model_layers_21_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_v_proj/mean":0.001331329345703125,"train/train/layer__model_layers_50/param/mean":0.0015899372547167512,"train/train/layer_model_layers_36/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_v_proj/std":0.2949218778803155,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/max_abs":0.1025390625,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/max_abs":0.0067138671875,"train/train/tensor_act_model_layers_51/mean":-0.0089111328125,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_23/grad/norm":0.031194762460447396,"train/train/tensor_param_model_layers_93_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/mean":0.000270843505859375,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/max_abs":0.00019550323486328125,"train/train/tensor_act_model_layers_54_self_attn_q_proj/norm":5670.885189653639,"train/train/tensor_act_model_layers_76_mlp_down_proj/mean":0.0019989013671875,"train/train/tensor_param_model_layers_15_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_k_proj/mean":-0.00388336181640625,"train/train/tensor_act_model_layers_15_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/max_abs":0.00019550323486328125,"train/train/tensor_act_model_layers_9_self_attn/std":0.0692176483655624,"train/train/tensor_act_model_layers_62_self_attn_v_proj/norm":2162.9267552195515,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/norm":4.34375,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/std":1.1275704970214918e-05,"train/train/layer_model_layers_29/grad/std":4.945236753481969e-05,"train/train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/max_abs":0.000637054443359375,"train/train/tensor_param_model_layers_28_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_89_mlp_gate_proj_weight/max_abs":0.0006103515625,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/norm":3.46875,"train_runtime":3361.0516,"train/train/layer__model_layers_35/param/std":0.05116893334625897,"train/train/tensor_act_model_layers_15_mlp/mean":-0.0004715919494628906,"train/train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_input_layernorm/std":1.0000004675238232,"train/train/layer_model_layers_92/act/mean":0.016256604875837053,"train/train/tensor_param_model_layers_3_mlp_gate_proj_weight/max_abs":0.10009765625,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/std":0.0001283763974008079,"train/train/tensor_act_model_layers_62_self_attn_q_proj/mean":0.0362548828125,"train/train/layer__model_layers_19/param/max_abs":1,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/mean":1.900480128824711e-08,"train/train/tensor_act_model_layers_93_mlp_gate_proj/std":0.8642603226264235,"train/train/layer_model_layers_5/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/norm":4.15625,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/norm":0.001433504153136152,"train/train/layer_model_layers_16/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/norm":0.014961147356678008,"train/train/layer_model_layers_61/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/norm":5.5625,"train/train/tensor_act_model_layers_42_mlp_down_proj/mean":0.0002837181091308594,"train/train/tensor_act_model_layers_26_mlp/norm":225.82528829992927,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/max_abs":0.0004291534423828125,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/norm":0.011657035767913755,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/norm":0.0013094472770441986,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/std":0.028076171875,"train/train/tensor_param_model_layers_28_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_8_mlp/std":0.04138237862212199,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/norm":6.28125,"train/train/tensor_act_model_layers_68_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer_model_layers_37/act/max_abs":9.3125,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/norm":4.21875,"train/train/tensor_act_model_layers_66_self_attn_o_proj/mean":0.003326416015625,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_21/act/mean":0.0008107934679303851,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/max_abs":0.1337890625,"train/train/layer__model_layers_23/param/max_abs":1,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_mlp_gate_proj_weight/norm":0.01545365078285221,"train/train/tensor_act_model_layers_23/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/std":4.143409014580988e-05,"train/train/tensor_param_model_layers_13_mlp_gate_proj_weight/norm":4.65625,"train/train/tensor_act_model_layers_83_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_q_proj/mean":0.0092010498046875,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_q_proj/mean":-0.02484130859375,"train/train/tensor_act_model_layers_88_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18/std":1.271490520031754,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/max_abs":0.000469207763671875,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/std":0.034912109375,"train/train/tensor_act_model_layers_41_self_attn_o_proj/max_abs":1.5546875,"train/train/tensor_act_model_layers_64_mlp_up_proj/norm":3323.4415153735695,"train/train/tensor_act_model_layers_27_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/std":0.036376953125,"train/train/tensor_act_model_layers_74_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/mean":8.782371878623962e-07,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/mean":-2.1889805793762207e-05,"train/train/tensor_grad_model_layers_60_mlp_gate_proj_weight/std":6.88282517734206e-05,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/max_abs":0.25,"train/train/layer__model_layers_35/param/norm":20.735481790354548,"train/train/tensor_grad_model_layers_42_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn/std":0.08180223172145651,"train/train/tensor_act_model_layers_29_mlp_gate_proj/max_abs":1.9296875,"train/train/tensor_act_model_layers_56_mlp/mean":-0.00019252300262451172,"train/train/tensor_act_model_layers_34_post_attention_layernorm/norm":5792.61376953375,"train/train/tensor_act_model_layers_36_self_attn_v_proj/mean":0.00681304931640625,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/std":3.9769842370036035e-05,"train/train/layer__model_layers_19/param/mean":0.0016444119945889144,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/mean":1.1444091796875e-05,"train/train/tensor_act_model_layers_18_mlp/std":0.03790298349009305,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_mlp_gate_proj_weight/std":5.74047319961064e-05,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/mean":0.0003509521484375,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/max_abs":0.1533203125,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_gate_proj/norm":5130.631103547445,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn/norm":513.9528027189612,"train/train/tensor_act_model_layers_53_self_attn_v_proj/norm":2197.468047176164,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/std":4.1092021839021896e-05,"train/train/tensor_act_model_layers_21_mlp_down_proj/mean":0.0017547607421875,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_14/act/norm":13934.346446152547,"train/train/tensor_grad_model_layers_32_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_k_proj/max_abs":4.53125,"train/train/tensor_act_model_layers_45_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/mean":7.584458217024803e-08,"train/train/layer_model_layers_34/act/std":0.6244607586339425,"train/train/tensor_act_model_layers_78/norm":9767.98811297993,"train/train/tensor_act_model_layers_17_mlp_up_proj/max_abs":2.015625,"train/train/tensor_act_model_layers_27_mlp_up_proj/mean":-0.002460479736328125,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/norm":0.030645362351612873,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/std":2.4864896838673188e-05,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/max_abs":0.000240325927734375,"train/train/tensor_param_model_layers_9_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/max_abs":0.0001373291015625,"train/train/tensor_grad_model_layers_71_mlp_gate_proj_weight/mean":-1.7695128917694092e-08,"train/train/tensor_act_model_layers_86_self_attn_v_proj/norm":2786.4421130199466,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/max_abs":0.00022125244140625,"train/train/tensor_act_model_layers_41_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_56/grad/std":6.404476250986833e-05,"train/train/tensor_act_model_layers_60_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_32_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/std":3.879883516601396e-05,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_mlp_gate_proj_weight/max_abs":0.00066375732421875,"train/train/layer_model_layers_51/act/norm":13648.849695856965,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/mean":-7.677590474486351e-08,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/norm":6.53125,"train/train/tensor_act_model_layers_86/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/max_abs":0.00156402587890625,"train/train/tensor_act_model_layers_83/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp_up_proj/std":0.3007812768692598,"train/train/tensor_act_model_layers_17_self_attn_o_proj/std":0.05798357486060793,"train/train/tensor_act_model_layers_55_mlp_down_proj/norm":469.1437218968507,"train/train/tensor_act_model_layers_38_self_attn_o_proj/mean":-2.3156404495239258e-05,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/norm":0.028720851406219196,"train/train/tensor_act_model_layers_66_mlp/norm":710.4658632805782,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/mean":-2.4621840566396713e-08,"train/train/tensor_act_model_layers_29_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_k_proj/mean":0.0045318603515625,"train/train/tensor_act_model_layers_54_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/std":0.0390625,"train/train/tensor_act_model_layers_4_self_attn_q_proj/std":1.1718750425179791,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/mean":9.514391422271729e-06,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/max_abs":0.197265625,"train/train/tensor_act_model_layers_59/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp_down_proj/norm":263.7055901079704,"train/train/global/grad/std":7.957094607276684e-05,"train/train/layer_model_layers_10/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/max_abs":0.001739501953125,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/std":0.0306396484375,"train/train/tensor_act_model_layers_62_mlp_up_proj/mean":0.004669189453125,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/std":0.0260009765625,"train/train/tensor_act_model_layers_82_self_attn/max_abs":2.46875,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/std":6.343362743135569e-05,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/std":8.42054585612224e-05,"train/train/layer_model_layers_92/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/norm":0.00675640502230449,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_post_attention_layernorm/std":1.0000021226783453,"train/train/tensor_act_model_layers_41_post_attention_layernorm/norm":5792.609008791248,"train/train/layer_model_layers_17/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_16/grad/std":5.7071834371409586e-05,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/mean":-8.440110832452774e-08,"train/train/tensor_act_model_layers_11/max_abs":8.625,"train/train/tensor_act_model_layers_14_mlp_up_proj/std":0.23217814383250474,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/max_abs":0.0004520416259765625,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/norm":7.25,"train/train/layer_model_layers_23/act/norm":13296.363417666164,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/std":0.0262451171875,"train/train/tensor_param_model_layers_60_mlp_gate_proj_weight/max_abs":0.1806640625,"train/train/tensor_act_model_layers_69_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/norm":0.03186432678445273,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/max_abs":0.00070953369140625,"eval/steps_per_second":3.994,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/max_abs":0.000408172607421875,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/max_abs":0.0005340576171875,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/norm":4.375,"train/train/tensor_param_model_layers_67_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/norm":0.001044414023068418,"train/train/layer_model_layers_80/grad/max_abs":0.00133514404296875,"train/train/tensor_act_model_layers_59_mlp/norm":579.4086061424512,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/norm":7.28125,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/mean":3.0919909477233887e-06,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/mean":0.0004024505615234375,"train/train/tensor_act_model_layers_19_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/max_abs":0.1533203125,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_2/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/norm":0.015789418608638427,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/max_abs":0.0003223419189453125,"train/train/layer__model_layers_28/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/mean":0.00011682510375976562,"train/train/tensor_act_model_layers_79_post_attention_layernorm/max_abs":5.96875,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14/max_abs":8.3125,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/std":0.0001134425634120659,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/std":6.646617283794923e-05,"train/train/tensor_act_model_layers_92/max_abs":18.125,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/std":4.816369786123827e-05,"train/train/layer__model_layers_6/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/mean":4.708999767899513e-08,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/norm":0.022984335794940552,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/std":0.0615234375,"train/train/tensor_act_model_layers_12_self_attn_o_proj/norm":442.59687073856117,"train/train/tensor_grad_model_layers_51_mlp_gate_proj_weight/max_abs":0.00063323974609375,"train/train/layer_model_layers_46/grad/norm":0.042546267521537964,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/mean":1.3620592653751373e-07,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/max_abs":0.000263214111328125,"train/train/tensor_act_model_layers_84_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/max_abs":0.0001983642578125,"train/train/layer_model_layers_66/grad/norm":0.06397298245122049,"train/train/tensor_act_model_layers_25_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_gate_proj_weight/mean":2.0477455109357834e-07,"train/train/tensor_act_model_layers_13_post_attention_layernorm/std":1.0000003813765215,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_90_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp/norm":206.20778720477077,"train/train/layer__model_layers_12/param/norm":19.96219571621569,"train/train/tensor_param_model_layers_46_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/max_abs":0.0006561279296875,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/std":9.06238227609261e-05,"train/train/tensor_act_model_layers_46_input_layernorm/norm":5792.608886722331,"train/train/tensor_act_model_layers_22/norm":7323.5855938846635,"train/train/tensor_act_model_layers_81_post_attention_layernorm/max_abs":5.78125,"train/train/tensor_act_model_layers_25_self_attn_k_proj/mean":0.04022216796875,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/mean":-4.458427429199219e-05,"train/train/tensor_param_model_layers_70_mlp_gate_proj_weight/std":0.03857421875,"train/train/tensor_act_model_layers_66_post_attention_layernorm/max_abs":5.5,"train/train/tensor_act_model_layers_92_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn/mean":-0.00015419721603393555,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_q_proj/norm":5744.532964078502,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/mean":0.00021839141845703125,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/max_abs":0.18359375,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/max_abs":0.000537872314453125,"train/train/tensor_act_model_layers_69_self_attn_k_proj/mean":-0.04364013671875,"train/train/tensor_act_model_layers_51_self_attn_k_proj/std":0.7666038531372447,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/norm":7.3125,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_78/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51/norm":7343.11071028625,"train/train/tensor_act_model_layers_70_post_attention_layernorm/mean":0.0077667236328125,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/mean":-0.00030517578125,"train/train/tensor_act_model_layers_44_input_layernorm/max_abs":6.40625,"train/train/tensor_act_model_layers_86_post_attention_layernorm/max_abs":5.84375,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/mean":-1.0251998901367188e-05,"train/train/tensor_act_model_layers_34_input_layernorm/norm":5792.612182617623,"train/train/tensor_act_model_layers_92_self_attn_o_proj/std":0.25391006229019114,"train/train/layer_model_layers_40/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/mean":-1.257285475730896e-06,"train/train/tensor_act_model_layers_8_self_attn_q_proj/norm":5677.80191851833,"train/train/tensor_act_model_layers_54_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp_down_proj/std":0.03759767018355074,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/max_abs":0.224609375,"train/train/layer__model_layers_50/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer__model_layers_90/param/mean":0.0017278852775204758,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_input_layernorm/mean":-0.031005859375,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/mean":2.6973430067300797e-07,"train/train/layer__model_layers_40/param/mean":0.0014692714173410687,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/max_abs":0.000782012939453125,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/std":0.02490234375,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/std":1.9983836548666942e-05,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/mean":-3.1322240829467773e-05,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/norm":7.625,"train/train/tensor_act_model_layers_93_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp_gate_proj/norm":5438.214836604279,"train/train/tensor_act_model_layers_31_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/norm":5.71875,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/norm":0.025484863870633985,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/mean":-0.000255584716796875,"train/train/tensor_act_model_layers_59_self_attn/max_abs":1.453125,"train/train/layer_model_layers_23/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/max_abs":0.0006561279296875,"train/train/tensor_act_model_layers_39_self_attn_o_proj/mean":-0.00208282470703125,"train/train/tensor_act_model_layers_78_mlp/max_abs":1.296875,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/mean":-0.00038909912109375,"train/train/tensor_param_model_layers_75_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/max_abs":0.0001392364501953125,"train/train/tensor_act_model_layers_84_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_o_proj/max_abs":0.453125,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/norm":0.0050214632375859105,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/norm":0.004737356408567599,"train/train/tensor_act_model_layers_8_mlp_gate_proj/norm":1839.3205733815084,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/norm":0.0007805042134448228,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/std":6.0152185599025355e-05,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/norm":5.1875,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_10/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_mlp_gate_proj_weight/mean":5.1975250244140625e-05,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp_gate_proj/norm":4279.559927571869,"train/train/tensor_param_model_layers_67_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn/max_abs":1.8359375,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/mean":5.373731255531311e-07,"train/train/tensor_act_model_layers_55_mlp_gate_proj/mean":-0.00604248046875,"train/train/tensor_act_model_layers_28_self_attn_v_proj/norm":2480.6219211322077,"train/train/tensor_param_model_layers_80_mlp_gate_proj_weight/mean":0.0003566741943359375,"train/train/tensor_act_model_layers_54_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/norm":0.0007806137407252417,"train/train/tensor_act_model_layers_7_self_attn_o_proj/norm":267.57110567920625,"train/train/layer_model_layers_82/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp/frac_near_user_limit":0,"train/train/layer__model_layers_24/param/mean":0.0015131672161417707,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/std":5.5873369850119286e-05,"train/train/tensor_act_model_layers_47_mlp/max_abs":0.53515625,"train/train/tensor_act_model_layers_54_self_attn_v_proj/mean":-0.0040435791015625,"train/train/tensor_act_model_layers_24_mlp_up_proj/mean":-0.0024871826171875,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_mlp_gate_proj_weight/norm":0.017552356906498305,"train/train/layer_model_layers_80/grad/std":8.577247843612009e-05,"train/train/tensor_act_model_layers_68_mlp_gate_proj/max_abs":2.828125,"train/train/tensor_act_model_layers_73_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/mean":-4.649162292480469e-05,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_41_mlp_gate_proj_weight/max_abs":0.15625,"train/train/tensor_act_model_layers_32_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/std":4.699304692690216e-05,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/mean":-6.532669067382812e-05,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/std":0.055908203125,"train/train/tensor_act_model_layers_46_mlp_gate_proj/max_abs":1.96875,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_q_proj/norm":6360.557381881519,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/mean":-4.2084138840436935e-08,"train/train/tensor_grad_model_layers_8_mlp_gate_proj_weight/mean":1.6526610124856234e-07,"train/train/tensor_param_model_layers_12_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_90/max_abs":13.25,"train/train/tensor_act_model_layers_61_mlp_down_proj/std":0.10351565774509885,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/mean":0.00031280517578125,"train/train/layer_model_layers_22/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61/max_abs":9.6875,"train/train/tensor_act_model_layers_15_mlp_up_proj/norm":1972.558296952659,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/mean":1.334119588136673e-07,"train/train/tensor_param_model_layers_87_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_0/act/std":0.8290860525524703,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/max_abs":0.142578125,"train/train/tensor_act_model_layers_76_self_attn_o_proj/mean":-0.0008754730224609375,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/max_abs":0.2373046875,"train/train/tensor_param_model_layers_33_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_23_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91/std":2.550792935587067,"train/train/layer_model_layers_20/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_o_proj/max_abs":1.2890625,"train/train/layer_model_layers_7/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/max_abs":0.1396484375,"train/train/tensor_act_model_layers_88_self_attn/std":0.1928742542415322,"train/train/tensor_act_model_layers_3_mlp_up_proj/norm":2256.5881468599464,"train/train/tensor_act_model_layers_50_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/norm":0.0353576402023698,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp_down_proj/mean":0.0008001327514648438,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/std":0.00013840802628427393,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/mean":5.163019523024559e-08,"train/train/layer__model_layers_29/param/max_abs":1,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/std":0.05126953125,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_2_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_35/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn/max_abs":0.66015625,"train/train/tensor_act_model_layers_78_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp_gate_proj/mean":0.00051116943359375,"train/train/layer__model_layers_4/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_q_proj/mean":0.0335693359375,"train/train/tensor_param_model_layers_78_mlp_gate_proj_weight/mean":0.0003757476806640625,"train/train/tensor_act_model_layers_64_self_attn/mean":0.00013637542724609375,"train/train/tensor_act_model_layers_71_self_attn_k_proj/std":0.8222689022578805,"train/train/tensor_param_model_layers_33_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_64_post_attention_layernorm/max_abs":5.6875,"train/train/tensor_act_model_layers_78_self_attn_v_proj/norm":2427.196942792348,"train/train/tensor_act_model_layers_35_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/std":0.03466796875,"train/train/tensor_param_model_layers_69_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_gate_proj_weight/max_abs":0.130859375,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/max_abs":0.1455078125,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/max_abs":0.333984375,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/mean":-7.963180541992188e-05,"train/train/tensor_grad_model_layers_4_mlp_gate_proj_weight/std":5.47140933718682e-05,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/std":0.046142578125,"train/train/tensor_param_model_layers_41_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_66_mlp_gate_proj_weight/std":0.0380859375,"train/train/tensor_grad_model_layers_7_mlp_gate_proj_weight/std":4.0667403993375695e-05,"train/train/tensor_act_model_layers_10_self_attn_o_proj/std":0.06006193038923408,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/std":0.0001226264366707038,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_67/act/norm":14720.831105413341,"train/train/tensor_act_model_layers_3_mlp_gate_proj/mean":-0.012176513671875,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/std":0.03125,"train/train/tensor_act_model_layers_91_mlp_gate_proj/mean":-0.0065460205078125,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/max_abs":0.000286102294921875,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/std":8.114105348004103e-05,"train/train/layer__model_layers_90/param/norm":26.32676647088472,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/std":4.508030539195696e-05,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/mean":-3.316745278425515e-08,"train/train/layer_model_layers_44/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/norm":0.0019934521721068827,"train/train/tensor_act_model_layers_34_post_attention_layernorm/std":1.0000007947670864,"train/train/tensor_act_model_layers_53_self_attn/max_abs":1.7578125,"train/train/tensor_act_model_layers_48_mlp/max_abs":0.51171875,"train/train/layer_model_layers_64/grad/mean":3.2988578871520186e-09,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/norm":0.004927596709830372,"train/train/tensor_grad_model_layers_30_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_input_layernorm/max_abs":6.15625,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_input_layernorm/norm":5792.606689458985,"train/train/tensor_act_model_layers_91_self_attn/mean":0.0044097900390625,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_82/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/norm":0.0067461796784353874,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/mean":9.261071681976318e-06,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/norm":7.125,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/std":3.776090671420218e-05,"train/train/tensor_grad_model_layers_24_mlp_gate_proj_weight/max_abs":0.000560760498046875,"train/train/tensor_act_model_layers_64_mlp_down_proj/norm":619.0999122773325,"train/train/tensor_grad_model_layers_24_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/max_abs":0.00012493133544921875,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/mean":-6.263144314289093e-08,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/mean":0.000537872314453125,"train/train/tensor_act_model_layers_58_self_attn_v_proj/mean":-0.000253528356552124,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/std":0.044189453125,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/mean":-1.8535356502979994e-06,"train/train/tensor_act_model_layers_8_self_attn_o_proj/std":0.04235904330503404,"train/train/layer__model_layers_59/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/max_abs":0.0003490447998046875,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/norm":0.002391325369635187,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_92/param/max_abs":1,"train/train/tensor_act_model_layers_50_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/norm":0.01394950451073474,"train/train/tensor_act_model_layers_59_mlp/std":0.10009766675350086,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/std":6.293348014273007e-05,"train/train/tensor_act_model_layers_12_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/max_abs":0.125,"train/train/tensor_param_model_layers_91_mlp_gate_proj_weight/max_abs":0.2451171875,"train/train/layer_model_layers_17/grad/std":4.931560671752042e-05,"train/train/tensor_act_model_layers_17_self_attn_k_proj/max_abs":4.3125,"train/train/tensor_param_model_layers_68_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/norm":0.0017402965002605388,"train/train/tensor_act_model_layers_76_self_attn/mean":-0.0008754730224609375,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/std":7.07455599112363e-05,"train/train/tensor_act_model_layers_2_mlp_down_proj/max_abs":1.5546875,"train/train/tensor_act_model_layers_51_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp_gate_proj/mean":-0.00739288330078125,"train/train/tensor_act_model_layers_65_self_attn/norm":954.5216318918541,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/max_abs":0.216796875,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/norm":0.006036505008338228,"train/train/tensor_act_model_layers_4_self_attn_v_proj/max_abs":2.390625,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/std":6.712000328477063e-05,"train/train/tensor_act_model_layers_35_self_attn_k_proj/max_abs":5.40625,"train/train/layer__model_layers_81/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_down_proj/mean":0.000980377197265625,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/max_abs":0.1298828125,"train/train/tensor_grad_model_layers_15_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/mean":-1.5333294868469238e-05,"train/train/tensor_act_model_layers_73_mlp/norm":829.1168934272308,"train/train/tensor_act_model_layers_40_mlp_up_proj/std":0.31250004023313266,"train/train/tensor_act_model_layers_43_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/mean":0.000278472900390625,"train/train/layer_model_layers_45/grad/std":4.461657417814024e-05,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/std":2.460429134342719e-05,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/norm":0.001176987117174542,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_post_attention_layernorm/max_abs":5.5625,"train/train/tensor_act_model_layers_83_mlp_down_proj/mean":0.0030975341796875,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/mean":8.771894499659538e-08,"train/train/tensor_grad_model_layers_91_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn/std":0.2241261923168629,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/std":7.130017347172732e-05,"train/train/tensor_grad_model_layers_74_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/mean":2.759043127298355e-08,"train/train/tensor_act_model_layers_60_self_attn_v_proj/max_abs":2.59375,"train/train/tensor_act_model_layers_80_input_layernorm/mean":0.0053253173828125,"train/train/tensor_act_model_layers_19_self_attn_k_proj/norm":4820.046059832434,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_k_proj/mean":-0.011566162109375,"train/train/tensor_grad_model_layers_60_mlp_gate_proj_weight/norm":0.024917747370993375,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/std":0.041015625,"train/train/tensor_act_model_layers_33_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/mean":-0.0004024505615234375,"train/train/tensor_act_model_layers_12_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_q_proj/norm":4842.100019652616,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_o_proj/std":0.06531000439404724,"train/train/tensor_act_model_layers_77_input_layernorm/std":1.0000019770503525,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/norm":7.6875,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_mlp_up_proj/norm":2672.7840910415503,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/std":7.773852676925611e-05,"train/train/layer_model_layers_13/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_up_proj/mean":-0.00319671630859375,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/max_abs":0.2265625,"train/train/tensor_act_model_layers_14_mlp_down_proj/std":0.03558366012116262,"train/train/tensor_act_model_layers_56_mlp_gate_proj/max_abs":2.515625,"train/train/tensor_param_model_layers_85_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_80/act/mean":0.011708259582519531,"train/train/tensor_act_model_layers_73_mlp_gate_proj/mean":0.0020904541015625,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/mean":0.0003376007080078125,"train/train/tensor_act_model_layers_60_mlp_gate_proj/max_abs":2.3125,"train/train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/mean":-1.4863908290863037e-05,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/norm":0.020264682974350943,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/std":0.050048828125,"train/train/tensor_act_model_layers_92_mlp/std":0.6425811083051888,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_27/grad/mean":1.9171878160812926e-07,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_45_self_attn_o_proj/mean":0.0008687973022460938,"train/train/tensor_act_model_layers_53_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_58/grad/std":6.0934307339874074e-05,"train/train/tensor_act_model_layers_86_self_attn_q_proj/max_abs":7,"train/train/tensor_act_model_layers_39_self_attn_k_proj/max_abs":4.96875,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/max_abs":0.251953125,"train/train/layer_model_layers_72/act/norm":15463.290789288749,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/norm":6.6875,"train/train/tensor_param_model_layers_85_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_91_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_input_layernorm/max_abs":5.09375,"train/train/tensor_grad_model_layers_65_mlp_gate_proj_weight/max_abs":0.000640869140625,"train/train/tensor_param_model_layers_30_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_down_proj/mean":-0.0043487548828125,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/std":8.586628781150565e-05,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/std":4.699870656429078e-05,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_q_proj/max_abs":6.1875,"train/train/tensor_param_model_layers_88_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/max_abs":0.0003108978271484375,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_o_proj/std":0.03820977964088249,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/max_abs":0.2333984375,"train/train/tensor_act_model_layers_31_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_input_layernorm/max_abs":6,"train/train/tensor_grad_model_layers_4_mlp_gate_proj_weight/norm":0.019794596070448724,"train/train/layer__model_layers_90/param/std":0.06493963981354202,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/std":0.00019136935247752163,"train/train/tensor_act_model_layers_11/norm":7450.350305205491,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp_down_proj/norm":1034.3043403417032,"train/train/tensor_param_model_layers_38_input_layernorm_weight/mean":1,"train/train/layer__model_layers_36/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_mlp_up_proj/norm":1978.1330431558324,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/norm":0.02018178918684697,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/max_abs":0.00090789794921875,"train/train/tensor_act_model_layers_61_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_gate_proj/norm":1956.0296236564418,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_65/act/max_abs":10.1875,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/mean":0.000225067138671875,"train/train/layer_model_layers_64/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn/norm":598.3049937312514,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_81_self_attn_q_proj/std":1.2832078164303746,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/std":0.0005233646751263214,"train/train/tensor_act_model_layers_58_self_attn_o_proj/std":0.09631812271311001,"train/train/tensor_act_model_layers_15_input_layernorm/max_abs":6,"train/train/tensor_act_model_layers_25_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/norm":6.46875,"train/train/layer__model_layers_28/param/max_abs":1,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/max_abs":0.0004711151123046875,"train/train/tensor_param_model_layers_18_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_mlp_down_proj/norm":829.1168934272308,"train/train/tensor_act_model_layers_30_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/norm":0.005675715757859835,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_86_self_attn/norm":1298.823011373343,"train/train/tensor_act_model_layers_50_self_attn/max_abs":1.640625,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/norm":5.59375,"train/train/tensor_act_model_layers_92_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_46_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_o_proj/norm":513.9528027189612,"train/train/tensor_param_model_layers_91_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_46/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/std":0.041259765625,"train/train/tensor_grad_model_layers_38_mlp_gate_proj_weight/norm":0.017615279032998175,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/std":0.0284423828125,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn_k_proj/mean":0.0419921875,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_k_proj/max_abs":4.625,"train/train/tensor_param_model_layers_1_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_25_self_attn/norm":459.95871946283614,"train/train/layer_model_layers_77/grad/std":8.417820724077846e-05,"train/train/layer__model_layers_31/param/max_abs":1,"train/train/tensor_act_model_layers_64_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/std":4.430264062098903e-05,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/max_abs":0.0004444122314453125,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/max_abs":0.0005950927734375,"train/train/tensor_act_model_layers_67_mlp/max_abs":0.95703125,"train/train/tensor_act_model_layers_38_post_attention_layernorm/std":1.0000004734610017,"train/train/tensor_act_model_layers_91_mlp_down_proj/norm":2993.8624633984227,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/norm":0.0046098971071185145,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_self_attn_v_proj/max_abs":2.484375,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_post_attention_layernorm/std":1.0000012456431677,"train/train/layer__model_layers_12/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/max_abs":0.134765625,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/mean":3.66847962141037e-06,"train/train/tensor_act_model_layers_18_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/mean":0.00019550323486328125,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/std":0.0269775390625,"train/train/tensor_grad_model_layers_71_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/mean":6.866455078125e-05,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/std":5.5947118972741074e-05,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/mean":3.2298266887664795e-06,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/mean":0.0001888275146484375,"train/train/layer__model_layers_63/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/mean":5.852780304849148e-08,"train/train/tensor_act_model_layers_34_post_attention_layernorm/mean":-0.0115203857421875,"train/train/tensor_act_model_layers_50_self_attn_q_proj/mean":-0.04595947265625,"train/train/tensor_act_model_layers_49_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_mlp_gate_proj_weight/std":4.143880436686012e-05,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/std":0.04345703125,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/mean":-0.000301361083984375,"train/train/tensor_param_model_layers_84_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/mean":-3.8463622331619263e-07,"train/train/tensor_act_model_layers_27_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/std":0.04052734375,"train/train/tensor_act_model_layers_27_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn_v_proj/norm":2134.6703922115566,"train/train/tensor_act_model_layers_29/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53/mean":-0.005096435546875,"train/train/tensor_act_model_layers_34_self_attn/norm":409.3614131422575,"train/train/tensor_grad_model_layers_80_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp/max_abs":1.2734375,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_gate_proj_weight/std":0.0299072265625,"train/train/tensor_act_model_layers_22_self_attn_q_proj/std":0.9892592962014763,"train/train/tensor_param_model_layers_41_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp_gate_proj/norm":1975.5074394781232,"train/train/tensor_act_model_layers_64_mlp_gate_proj/mean":-0.010284423828125,"train/train/tensor_act_model_layers_79_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_12/act/mean":-0.009665352957589286,"train/train/tensor_act_model_layers_84_self_attn/norm":1278.6258776078105,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/max_abs":0.134765625,"train/train/tensor_act_model_layers_33_self_attn_q_proj/max_abs":5.125,"train/train/tensor_act_model_layers_83_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/max_abs":0.0001773834228515625,"train/train/layer__model_layers_28/param/norm":20.94414540062449,"train/train/layer_model_layers_37/grad/max_abs":0.001739501953125,"train/train/tensor_act_model_layers_59_self_attn_k_proj/mean":0.05059814453125,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_post_attention_layernorm/std":1.000000879022605,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/norm":0.028174518332134035,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/norm":4.28125,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/norm":0.017182670261466942,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/norm":0.015537373884000715,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/std":0.0277099609375,"train/train/tensor_act_model_layers_10_mlp_down_proj/max_abs":0.458984375,"train/train/tensor_act_model_layers_73_self_attn_q_proj/std":0.9941464032378083,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/norm":0.02432608966974299,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_mlp_gate_proj_weight/max_abs":0.00074005126953125,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/mean":-7.904600352048874e-08,"train/train/tensor_act_model_layers_46_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_38/act/std":0.6415437901913839,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/mean":-0.0001888275146484375,"train/train/tensor_act_model_layers_16_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/norm":5.84375,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/mean":-6.16908073425293e-06,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/std":0.040283203125,"train/train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_o_proj/mean":0.000804901123046875,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/mean":0.0005035400390625,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_30/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_48/grad/norm":0.039634289936690666,"train/train/tensor_act_model_layers_30_mlp_up_proj/mean":0.00494384765625,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/mean":-6.628688424825668e-07,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/norm":0.023326531608111878,"train/train/tensor_act_model_layers_59_mlp/mean":-0.00052642822265625,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/norm":0.0030365852842107223,"train/train/tensor_act_model_layers_41_post_attention_layernorm/mean":-0.0137481689453125,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/max_abs":0.000270843505859375,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/max_abs":0.0002765655517578125,"train/train/tensor_act_model_layers_18_self_attn_v_proj/mean":-8.499622344970703e-05,"train/train/layer__model_layers_21/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/norm":0.001553935555023032,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/max_abs":0.2412109375,"train/train/tensor_act_model_layers_61_mlp/norm":599.8585135776389,"train/train/tensor_act_model_layers_85_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/max_abs":0.0004291534423828125,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn/mean":-0.00014060735702514648,"train/train/layer_model_layers_7/grad/std":4.8144881982410756e-05,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/mean":-3.91155481338501e-08,"train/train/tensor_act_model_layers_29_mlp_down_proj/norm":239.01728178106117,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/std":2.3580178852565355e-05,"train/train/tensor_act_model_layers_6_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28/norm":7225.481485536012,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/max_abs":0.2236328125,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/mean":2.2009015083312988e-05,"train/train/tensor_act_model_layers_78_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_k_proj/max_abs":5.15625,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/mean":0.0004558563232421875,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/norm":4.78125,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/max_abs":0.1357421875,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/mean":-4.095491021871567e-07,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/norm":5.09375,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp_gate_proj/std":0.23437502082281966,"train/train/tensor_act_model_layers_31_self_attn/mean":-0.00011797621846199036,"train/train/tensor_act_model_layers_49_self_attn_k_proj/std":0.7656251002026998,"train/train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_0/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/norm":3.5625,"train/train/tensor_act_model_layers_24_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_40/param/max_abs":1,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/max_abs":0.0001697540283203125,"train/train/tensor_act_model_layers_9_self_attn_o_proj/mean":-0.0003643035888671875,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/max_abs":0.00115966796875,"train/train/tensor_param_model_layers_43_mlp_gate_proj_weight/norm":5.625,"train/train/layer_model_layers_80/act/norm":17291.87273729365,"train/train/tensor_act_model_layers_53_input_layernorm/std":1.0000003892927603,"train/train/tensor_act_model_layers_26_self_attn_o_proj/std":0.07190223005910458,"train/train/tensor_param_model_layers_93_mlp_gate_proj_weight/mean":0.0004444122314453125,"train/train/tensor_grad_model_layers_83_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/norm":0.000764249630342452,"train/train/tensor_act_model_layers_55_mlp_gate_proj/std":0.36914065590611084,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/mean":-1.1095107765868306e-07,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/std":0.047119140625,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/norm":0.003146919474955129,"train/train/tensor_act_model_layers_46_mlp_up_proj/max_abs":2.03125,"train/train/layer_model_layers_28/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn_q_proj/norm":5428.984558638867,"train/train/tensor_act_model_layers_61_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/mean":-7.486343383789062e-05,"train/train/tensor_act_model_layers_41_mlp/norm":333.138692138353,"train/train/tensor_act_model_layers_45_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/norm":4.5625,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38/norm":7219.134342420796,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/max_abs":0.000896453857421875,"train/train/tensor_act_model_layers_28/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/mean":4.991888999938965e-07,"train/train/layer_model_layers_9/act/norm":13545.409779897464,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_52_input_layernorm/mean":-0.00606536865234375,"train/train/tensor_act_model_layers_71_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/std":0.036376953125,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/norm":6.0625,"train/train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/max_abs":0.000423431396484375,"train/train/tensor_act_model_layers_70_post_attention_layernorm/norm":5792.610473633269,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/norm":3.359375,"train/train/tensor_act_model_layers_49_mlp_gate_proj/norm":2799.0700716818646,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_4/grad/std":7.779491753618246e-05,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/max_abs":1,"train/train/layer__model_layers_27/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn_k_proj/mean":0.05780029296875,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_81_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_45/param/std":0.052133099920544976,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/mean":-2.1455343812704086e-07,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/norm":0.018747631559193942,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/max_abs":0.0003662109375,"train/train/tensor_act_model_layers_86_mlp/max_abs":2.03125,"train/train/tensor_act_model_layers_17_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp_down_proj/mean":0.0008840560913085938,"train/train/tensor_act_model_layers_90_mlp_down_proj/std":0.4199219106934776,"train/train/tensor_act_model_layers_29_mlp_gate_proj/mean":-0.001537322998046875,"train/train/tensor_act_model_layers_85_input_layernorm/std":1.0000009651698636,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/max_abs":0.0003204345703125,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/std":0.046142578125,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/mean":-0.0002689361572265625,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/mean":0.0002040863037109375,"train/train/tensor_act_model_layers_81_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_mlp/mean":0.0010747909545898438,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/std":0.05322265625,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/std":0.025146484375,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/norm":6.375,"train/train/tensor_param_model_layers_68_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/norm":0.002113535076946259,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/norm":4.09375,"train/train/tensor_act_model_layers_11_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/max_abs":0.10400390625,"train/train/tensor_act_model_layers_53_self_attn/mean":0.0024871826171875,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/max_abs":0.0001850128173828125,"train/train/tensor_act_model_layers_25/std":1.2636786328773275,"train/train/tensor_act_model_layers_74_self_attn_k_proj/mean":0.0408935546875,"train/train/tensor_act_model_layers_70_self_attn_o_proj/mean":0.0009250640869140625,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/std":2.753053424347658e-05,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/std":2.91822858141323e-05,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/max_abs":0.00011396408081054688,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/max_abs":0.1396484375,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/std":1.2506604530306302e-05,"train/train/tensor_act_model_layers_2_self_attn_o_proj/std":0.038760165551908504,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/max_abs":0.0003490447998046875,"train/train/tensor_param_model_layers_18_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/std":1.8505691246829084e-05,"train/train/tensor_param_model_layers_68_mlp_gate_proj_weight/mean":-0.0002880096435546875,"train/train/layer_model_layers_93/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_mlp_gate_proj_weight/mean":-2.4959444999694824e-07,"train/train/tensor_act_model_layers_28_mlp_down_proj/mean":0.0003266334533691406,"train/train/layer_model_layers_48/act/norm":13605.359095730297,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/std":0.05615234375,"train/train/layer_model_layers_37/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn/std":0.04235904330503404,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/norm":0.005896016628316499,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_input_layernorm/norm":5792.613891603676,"train/train/tensor_act_model_layers_41_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/norm":0.029208176173459846,"train/train/tensor_act_model_layers_93_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/mean":1.1995434761047363e-06,"train/train/tensor_param_model_layers_75_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/max_abs":0.255859375,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/std":0.0001472271617550287,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/max_abs":0.2197265625,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/std":1.0765198907521928e-05,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/norm":0.0006152404253147476,"train/train/tensor_act_model_layers_89_self_attn_q_proj/max_abs":6.6875,"train/train/tensor_act_model_layers_72_mlp_gate_proj/norm":3782.2992507307263,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/std":1.5884706712078475e-05,"train/train/layer__model_layers_5/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/std":7.510736904811051e-05,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/max_abs":0.255859375,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/std":5.682977819778805e-05,"train/train/tensor_act_model_layers_5_mlp_up_proj/max_abs":1.890625,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/max_abs":0.10302734375,"train/train/tensor_act_model_layers_55_post_attention_layernorm/max_abs":5.84375,"train/train/tensor_param_model_layers_25_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_45_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/mean":8.509960025548935e-08,"train/train/layer_model_layers_67/grad/mean":1.0504258009089322e-07,"train/train/tensor_act_model_layers_43_self_attn_o_proj/mean":0.0011234283447265625,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/std":3.8706481377223014e-05,"train/train/tensor_param_model_layers_64_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/mean":-5.029141902923584e-08,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_65/grad/std":8.194404656325652e-05,"train/train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_o_proj/mean":0.0006742477416992188,"train/train/tensor_act_model_layers_51_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85/max_abs":11.75,"train/train/layer__model_layers_39/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_19/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_input_layernorm/mean":-0.027008056640625,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/max_abs":0.189453125,"train/train/tensor_act_model_layers_23_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_/norm":4.001687029743988,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/mean":-3.282912075519562e-08,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/std":0.00011840152138062626,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/mean":5.737645551562309e-06,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_62/grad/max_abs":0.00113677978515625,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/norm":0.013211080366520541,"train/train/tensor_act_model_layers_40_mlp_gate_proj/norm":2564.150184273814,"train/train/tensor_act_model_layers_47_mlp_up_proj/max_abs":2.125,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/norm":0.005809351854293623,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/max_abs":0.0002498626708984375,"train/train/tensor_act_model_layers_91_self_attn_k_proj/norm":6521.383817248574,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/norm":0.0053685138748721,"train/train/tensor_param_model_layers_8_mlp_gate_proj_weight/std":0.0244140625,"train/train/tensor_act_model_layers_91/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/max_abs":0.00022792816162109375,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/norm":0.016710289971573977,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/max_abs":0.0005035400390625,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/norm":0.0323007481120783,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/max_abs":0.25390625,"train/train/tensor_act_model_layers_67_mlp/std":0.12060549789506718,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/max_abs":0.00055694580078125,"train/train/layer__model_layers_4/param/max_abs":1,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/std":7.201047474022276e-05,"train/train/tensor_act_model_layers_37_post_attention_layernorm/norm":5792.606933596374,"train/train/tensor_param_model_layers_88_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_mlp_gate_proj/mean":0.0139923095703125,"train/train/tensor_param_model_layers_29_mlp_gate_proj_weight/mean":-0.0002765655517578125,"train/train/tensor_act_model_layers_83_mlp/mean":0.0030975341796875,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/mean":-0.0004119873046875,"train/train/tensor_grad_model_layers_93_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/mean":-0.008331298828125,"train/train/tensor_act_model_layers_31_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_3/param/mean":0.0016076241194179018,"train/train/tensor_act_model_layers_55_self_attn_v_proj/std":0.3325207144132226,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/std":5.129816614421156e-05,"train/train/tensor_param_model_layers_22_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/mean":-2.477318048477173e-06,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/norm":4.8125,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_59/param/std":0.05612847128361815,"train/train/tensor_act_model_layers_44_self_attn_k_proj/mean":-0.001964569091796875,"train/train/layer_model_layers_51/grad/std":4.978536965549172e-05,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/max_abs":0.1669921875,"train/train/layer__model_layers_48/param/std":0.053661570132250154,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/max_abs":0.000362396240234375,"train/train/tensor_act_model_layers_64_self_attn/norm":428.1534843436557,"train/train/tensor_act_model_layers_82_input_layernorm/max_abs":5.625,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_mlp_gate_proj_weight/norm":7.84375,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/norm":8.1875,"train/train/tensor_act_model_layers_48_mlp_down_proj/max_abs":0.51171875,"train/train/tensor_param_model_layers_36_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_k_proj/std":0.7304690420866066,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/std":0.043212890625,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/mean":4.0605664253234863e-07,"train/train/tensor_act_model_layers_68_self_attn_v_proj/max_abs":5,"train/train/tensor_act_model_layers_10_self_attn_q_proj/std":0.8916032249979917,"train/train/tensor_act_model_layers_58/norm":7567.641009515676,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/std":4.1049343173714495e-05,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/mean":-4.130415618419647e-07,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/max_abs":0.140625,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/std":2.6113625810162943e-05,"train/train/layer_model_layers_0/grad/std":0.00040325340041363455,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_72/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/max_abs":0.13671875,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/mean":2.0714651327580214e-08,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/mean":0.0001659393310546875,"train/train/layer_model_layers_83/grad/norm":0.06609038216299248,"train/train/tensor_act_model_layers_84_mlp_down_proj/mean":-0.00225830078125,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_q_proj/std":0.9296876662919352,"train/train/layer_model_layers_91/act/std":1.0145663064113435,"train/train/layer_model_layers_1/grad/norm":0.08664489930765319,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/max_abs":0.0006256103515625,"train/train/tensor_param_model_layers_50_mlp_gate_proj_weight/std":0.032470703125,"train/train/tensor_param_model_layers_65_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_68/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_input_layernorm/mean":0.01141357421875,"train/train/layer_model_layers_18/grad/std":4.213336337891727e-05,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_33/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_65/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_37/grad/mean":1.4115711921090827e-07,"train/train/tensor_act_model_layers_73_self_attn_v_proj/max_abs":3.078125,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/std":1.0129916713993505e-05,"train/train/tensor_act_model_layers_71_input_layernorm/mean":0.00742340087890625,"train/train/tensor_act_model_layers_64_self_attn/std":0.0738554636007003,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/std":0.02490234375,"train/train/tensor_act_model_layers_66_self_attn/norm":1158.0701733151552,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_post_attention_layernorm/norm":5792.609130864329,"train/train/tensor_act_model_layers_84_self_attn_o_proj/mean":-0.0010423660278320312,"train/train/tensor_act_model_layers_75_mlp/std":0.15625006876651548,"train/train/tensor_act_model_layers_4_post_attention_layernorm/mean":-0.02978515625,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_lm_head/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_gate_proj_weight/mean":-5.27501106262207e-06,"train/train/layer_model_layers_28/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_8/grad/mean":-1.3670536824581962e-08,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/mean":-0.00018405914306640625,"train/train/tensor_act_model_layers_75_input_layernorm/norm":5792.615844728295,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_o_proj/mean":-0.0010061264038085938,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/std":3.569099786526401e-05,"train/train/layer_model_layers_84/act/max_abs":11.5,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/std":7.094292729603743e-05,"train/train/tensor_act_model_layers_82_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/max_abs":0.1474609375,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/max_abs":0.2177734375,"train/train/tensor_act_model_layers_57_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/std":4.286438710520837e-05,"train/train/tensor_act_model_layers_16_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80/std":1.7558654097700745,"train/train/tensor_param_model_layers_51_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_29_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/norm":0.013158562527199212,"train/train/tensor_act_model_layers_89_mlp_down_proj/norm":2141.399245173063,"train/train/layer_model_layers_7/act/max_abs":8.75,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_k_proj/mean":-0.082275390625,"train/train/tensor_act_model_layers_53_mlp/frac_near_dtype_limit":0,"train/train/layer__model_layers_46/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/std":5.300515865544182e-05,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_mlp_gate_proj_weight/mean":6.532669067382812e-05,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/norm":3.3125,"train/train/tensor_act_model_layers_35_self_attn/norm":228.78234637280647,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/std":0.043701171875,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/std":0.0390625,"train/train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_q_proj/max_abs":5.3125,"train/train/tensor_act_model_layers_50_self_attn_o_proj/norm":541.1735998029751,"train/train/tensor_param_model_layers_39_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp/max_abs":0.341796875,"train/train/tensor_act_model_layers_19_mlp_up_proj/std":0.24609397880958903,"train/train/layer_model_layers_53/grad/norm":0.04430039423859639,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_post_attention_layernorm/mean":0.0010080337524414062,"train/train/tensor_act_model_layers_92_input_layernorm/mean":0.0132598876953125,"train/train/tensor_act_model_layers_28_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/std":6.354028246615963e-05,"train/train/layer_model_layers_57/grad/max_abs":0.0021514892578125,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_o_proj/std":0.2448744342602649,"train/train/tensor_act_model_layers_41_self_attn_k_proj/std":0.7998065430290114,"train/train/tensor_act_model_layers_4/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/mean":4.233792424201965e-06,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/std":4.353481289468411e-05,"train/train/tensor_param_model_layers_61_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_60_post_attention_layernorm/std":1.0000006140774464,"train/train/tensor_act_model_layers_58_self_attn_q_proj/std":0.9365292132282064,"train/train/tensor_act_model_layers_50_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/max_abs":0.00037384033203125,"train/train/tensor_act_model_layers_40_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/max_abs":0.1318359375,"train/train/tensor_act_model_layers_85_self_attn/norm":1110.1296840109146,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/mean":-7.525086402893066e-07,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/norm":0.025911274946065408,"train/train/tensor_param_model_layers_44_input_layernorm_weight/std":0,"train/train/layer_model_layers_27/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_post_attention_layernorm/std":1.0000014131645931,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/max_abs":0.255859375,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/max_abs":0.0003719329833984375,"train/train/tensor_param_model_layers_86_mlp_gate_proj_weight/std":0.046875,"train/train/tensor_act_model_layers_46_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_77/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/max_abs":5.555152893066406e-05,"train/train/layer__model_layers_26/param/std":0.05074205607593157,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/std":0.00010320768855088198,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/mean":4.872679710388184e-06,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/mean":0.0001316070556640625,"train/train/tensor_act_model_layers_6/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/std":4.043546633372209e-05,"train/train/tensor_param_model_layers_77_mlp_gate_proj_weight/mean":0.00018787384033203125,"train/train/tensor_act_model_layers_89_mlp_gate_proj/mean":-0.00521087646484375,"train/train/tensor_act_model_layers_71/max_abs":11,"train/train/tensor_act_model_layers_22_self_attn_q_proj/max_abs":6.8125,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/norm":0.02519369785546651,"train/train/tensor_act_model_layers_24_self_attn/norm":162.69204488932456,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_mlp_gate_proj_weight/max_abs":0.16796875,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/norm":7.21875,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/std":0.0311279296875,"train/train/tensor_act_model_layers_14_mlp_gate_proj/std":0.2324218940334152,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/std":2.4088344101360192e-05,"train/train/tensor_act_model_layers_20_self_attn_q_proj/std":0.8076193903005363,"train/train/tensor_act_model_layers_23_self_attn/std":0.04907457551704344,"train/train/tensor_param_model_layers_1_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_mlp_gate_proj_weight/std":0.0296630859375,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_gate_proj_weight/mean":1.5341356629505754e-07,"train/train/tensor_act_model_layers_22_self_attn/std":0.051271743515113787,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_55/param/std":0.05293953035599746,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/norm":0.015310611462423393,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/std":0.0299072265625,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn/norm":400.6547887434907,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/std":0.0257568359375,"train/train/tensor_param_model_layers_59_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/std":9.84134762899329e-05,"train/train/tensor_param_model_layers_66_mlp_gate_proj_weight/max_abs":0.1630859375,"train/train/tensor_act_model_layers_55_mlp_up_proj/std":0.37109390839832934,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_38/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer__model_layers_66/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_post_attention_layernorm/max_abs":6.3125,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn/std":0.06006193038923408,"train/train/tensor_act_model_layers_34_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_mlp_down_proj/norm":678.5794164842906,"train/train/tensor_act_model_layers_4_mlp_up_proj/max_abs":2.453125,"train/train/tensor_param_model_layers_55_mlp_gate_proj_weight/max_abs":0.1630859375,"train/train/tensor_act_model_layers_80_post_attention_layernorm/mean":0.005458831787109375,"train/train/tensor_act_model_layers_22_self_attn_o_proj/std":0.051271743515113787,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/norm":6.71875,"train/train/tensor_act_model_layers_60_self_attn_v_proj/norm":2280.282042837745,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_0/act/max_abs":8.5625,"train/train/tensor_act_model_layers_54_self_attn/std":0.13940672256917103,"train/train/tensor_act_model_norm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn_v_proj/std":0.4150403137047925,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/mean":-4.796311259269714e-07,"train/train/tensor_act_model_layers_27_mlp/std":0.03985612114906112,"train/train/tensor_grad_model_layers_74_mlp_gate_proj_weight/std":7.452834601383622e-05,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_71_mlp_gate_proj/max_abs":2.6875,"train/train/layer_model_layers_37/act/frac_near_user_limit":0,"train/train/layer_model_layers_79/act/mean":-0.0004356588636125837,"train/train/tensor_act_model_layers_60_self_attn_q_proj/mean":-0.077392578125,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/mean":-1.996522769331932e-08,"train/train/layer_model_layers_77/grad/norm":0.0682490430205846,"train/train/layer__model_layers_43/param/std":0.05363406616381998,"train/train/tensor_act_model_layers_75_mlp_up_proj/max_abs":2.6875,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/mean":4.498288035392761e-07,"train/train/tensor_act_model_layers_2_mlp_up_proj/max_abs":2.4375,"train/train/tensor_act_model_layers_26_self_attn/norm":416.3950798033205,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_down_proj/max_abs":0.3515625,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/norm":0.017800409162029903,"train/train/layer_model_layers_70/grad/norm":0.05963190611926263,"train/train/tensor_act_model_layers_23/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/norm":0.03324774941019618,"train/train/tensor_act_model_layers_88_input_layernorm/mean":0.004486083984375,"train/train/tensor_act_model_layers_34_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/norm":0.0005712310137967687,"train/train/tensor_param_model_layers_68_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_90/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_30/param/max_abs":1,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/mean":-1.1315569281578064e-07,"train/train/tensor_param_model_layers_92_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/max_abs":0.00030517578125,"train/train/tensor_act_model_layers_23_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_mlp_gate_proj_weight/mean":1.6350531950592995e-07,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/mean":-9.099021553993225e-07,"train/train/layer_model_layers_38/grad/max_abs":0.00145721435546875,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_k_proj/std":1.0156251320472045,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/mean":-0.0001621246337890625,"train/train/tensor_act_model_layers_86_mlp_down_proj/max_abs":2.03125,"train/train/layer_model_layers_8/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_post_attention_layernorm/mean":0.004863739013671875,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_k_proj/max_abs":5.28125,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_self_attn_v_proj/max_abs":2.671875,"train/train/tensor_param_model_layers_19_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_48_mlp_gate_proj_weight/max_abs":0.0004787445068359375,"train/train/layer__model_layers_37/param/norm":21.68313176469211,"train/train/layer_model_layers_46/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/norm":0.027899543127127065,"train/train/layer_model_layers_66/grad/std":7.90001294436641e-05,"train/train/layer__model_layers_32/param/norm":21.05019229610801,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/max_abs":0.00078582763671875,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/max_abs":0.00030517578125,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/std":0.039794921875,"train/train/tensor_act_model_layers_81_mlp/mean":-0.0003800392150878906,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/std":0.032958984375,"train/train/tensor_param_model_layers_34_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/max_abs":0.1572265625,"train/train/tensor_grad_model_layers_69_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/mean":1.6435980796813965e-05,"train/train/tensor_param_model_layers_21_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_44_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/std":5.878920232213389e-05,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/mean":-0.00010442733764648438,"train/train/tensor_grad_model_layers_56_mlp_gate_proj_weight/std":6.154635530918642e-05,"train/train/tensor_act_model_layers_21_self_attn/mean":0.0013284683227539062,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/max_abs":0.1083984375,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/std":0.00017305041699546494,"train/train/tensor_act_model_layers_87_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/std":1.7375001158103814e-05,"train/train/tensor_param_model_layers_9_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_input_layernorm/norm":5792.606201177663,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/max_abs":0.1669921875,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_lm_head/max_abs":17.5,"train/train/tensor_act_model_layers_28_mlp_up_proj/norm":2146.693669288433,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/mean":-6.818771362304688e-05,"train/train/tensor_act_model_layers_12_mlp_gate_proj/max_abs":1.84375,"train/train/tensor_act_model_layers_60_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_gate_proj_weight/max_abs":0.1611328125,"train/train/tensor_act_model_layers_15_mlp/std":0.03973478515877061,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_32_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/mean":1.2293457984924316e-07,"train/train/tensor_act_model_layers_78_input_layernorm/norm":5792.606201173465,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/mean":-0.0002002716064453125,"train/train/layer_model_layers_42/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/norm":0.020113746571006244,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/std":0.0001571764216455202,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/norm":0.009136842613473336,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/std":4.7099486386599634e-05,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_66/act/max_abs":10.5,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/std":6.640869363788014e-05,"train/train/tensor_act_model_layers_26_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/max_abs":0.00084686279296875,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/max_abs":0.11572265625,"train/train/tensor_act_model_layers_89_input_layernorm/mean":0.00812530517578125,"train/train/tensor_act_model_layers_36_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/std":0.0208740234375,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn_v_proj/std":0.36035319209376465,"train/train/tensor_param_model_layers_62_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_7_self_attn_v_proj/mean":-0.0003147721290588379,"train/train/tensor_act_model_layers_73_self_attn_o_proj/std":0.12281141495276253,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_45/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/mean":-8.249282836914062e-05,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/std":7.579972137350457e-05,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/mean":-9.918212890625e-05,"train/train/tensor_act_model_layers_77/mean":0.0032196044921875,"train/train/tensor_act_model_layers_64_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_post_attention_layernorm/max_abs":5.8125,"train/train/tensor_grad_model_layers_68_mlp_gate_proj_weight/mean":6.105983629822731e-08,"train/train/tensor_act_model_layers_90_mlp_down_proj/norm":2434.5026698431375,"train/global_step":1500,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/std":0.00010040056592999202,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/norm":0.010881602982060435,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp/mean":7.265806198120117e-05,"train/train/tensor_param_model_layers_86_mlp_gate_proj_weight/norm":8.5,"train/train/tensor_param_model_layers_32_mlp_gate_proj_weight/std":0.0281982421875,"train/train/tensor_act_model_layers_9_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn/std":0.18750207811763583,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/mean":2.2724270820617676e-07,"train/train/tensor_act_model_layers_26_self_attn_q_proj/std":0.9345719595361963,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/norm":0.010041813603324666,"train/train/tensor_act_model_layers_34_mlp_up_proj/std":0.2851562936828528,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/max_abs":0.00017452239990234375,"train/train/tensor_act_model_layers_65_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/norm":1778.6654644670336,"train/train/tensor_grad_model_layers_53_mlp_gate_proj_weight/mean":-4.8603396862745285e-09,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/norm":4.71875,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/mean":0.0001354217529296875,"train/train/tensor_act_model_layers_76_self_attn_o_proj/max_abs":1.65625,"train/train/tensor_param_model_layers_42_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_12_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/std":2.1989468031966806e-05,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/mean":-6.565824151039124e-08,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/mean":1.9556318875402212e-07,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_81_mlp_down_proj/norm":1196.4930190277144,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/max_abs":0.00145721435546875,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/std":0.037841796875,"train/train/tensor_act_model_layers_69_self_attn_v_proj/max_abs":3.09375,"train/train/layer__model_layers_42/param/max_abs":1,"train/train/tensor_act_model_layers_93_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_gate_proj/std":0.6640633428792653,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/max_abs":0.0003681182861328125,"train/train/tensor_act_model_layers_52_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/std":0.04248046875,"train/train/tensor_param_model_layers_8_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/std":0.043212890625,"train/train/tensor_act_model_layers_13_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/std":0.027099609375,"train/train/layer__model_layers_67/param/std":0.0569439674145603,"train/train/layer_model_layers_52/act/std":0.6241815398533933,"train/train/layer_model_layers_84/act/std":0.8451400368525857,"train/train/tensor_act_model_layers_14_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_24_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/norm":4.6875,"train/train/tensor_act_model_layers_26_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_mlp_gate_proj_weight/std":0.0281982421875,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/std":0.029052734375,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_v_proj/std":0.49414142882551687,"train/train/tensor_act_model_layers_85_mlp/std":0.2495122097004952,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/std":0.04052734375,"train/train/layer_model_layers_75/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91/mean":0.028717041015625,"train/train/tensor_param_model_layers_0_mlp_gate_proj_weight/norm":7.3125,"train/train/layer__model_layers_5/param/std":0.04844503029842731,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/std":5.719267621176688e-05,"train/train/layer_model_layers_10/grad/max_abs":0.00145721435546875,"train/train/layer_model_layers_61/act/mean":0.002368963190487453,"train/train/layer__model_layers_11/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp_gate_proj/norm":3020.7134445103793,"train/train/tensor_act_model_layers_77_mlp/norm":1016.727593024412,"train/train/tensor_act_model_layers_71_mlp_up_proj/mean":0.0073394775390625,"train/train/tensor_act_model_layers_81_mlp_gate_proj/mean":0.00470733642578125,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/mean":-0.00024318695068359375,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/mean":-7.43865966796875e-05,"train/train/tensor_grad_model_layers_35_mlp_gate_proj_weight/std":4.6163324274909325e-05,"train/train/tensor_act_model_layers_47_input_layernorm/max_abs":6.25,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/mean":1.371372491121292e-07,"train/train/layer_model_layers_26/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/norm":0.017340215718094597,"train/train/layer_model_layers_83/act/mean":0.00989396231515067,"train/train/tensor_grad_model_layers_28_mlp_gate_proj_weight/max_abs":0.0006256103515625,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/std":0.044189453125,"train/train/tensor_act_model_layers_61_self_attn/mean":0.0003173351287841797,"train/train/tensor_param_model_layers_31_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_post_attention_layernorm/max_abs":5.625,"train/train/tensor_act_model_layers_11_mlp_down_proj/std":0.04101672365709403,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/max_abs":0.000598907470703125,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/max_abs":0.000476837158203125,"train/train/tensor_act_model_layers_23_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn/max_abs":1.0546875,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/mean":-0.00010204315185546875,"train/train/tensor_act_model_layers_55_mlp_down_proj/std":0.08105471055584082,"train/train/tensor_param_model_layers_68_mlp_gate_proj_weight/max_abs":0.1630859375,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_o_proj/max_abs":2.625,"train/train/tensor_param_model_layers_13_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn_k_proj/norm":4017.744627138954,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/norm":0.01566203020948503,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/std":0.034912109375,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp/norm":491.13860279983805,"train/train/tensor_act_model_layers_1_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_down_proj/max_abs":0.53125,"train/train/tensor_act_model_layers_33_mlp_gate_proj/norm":2313.8505146782913,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/max_abs":0.2421875,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/mean":-1.9595609046518803e-07,"train/train/tensor_param_model_layers_42_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_mlp_gate_proj_weight/norm":5.75,"train/train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/norm":0.0008269428724814287,"train/train/tensor_grad_model_layers_72_mlp_gate_proj_weight/max_abs":0.000507354736328125,"train/train/tensor_param_model_layers_23_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_0_self_attn_k_proj/norm":2537.793907623176,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/std":0.02685546875,"train/train/layer_model_layers_61/grad/mean":-1.7077128828594726e-07,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/mean":0.0007476806640625,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/norm":5.25,"train/train/layer_model_layers_32/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43/frac_near_user_limit":0,"train/train/layer__model_layers_30/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/mean":2.1644154912792146e-08,"train/train/tensor_param_model_layers_16_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn/std":0.07947030556452747,"train/train/tensor_act_model_layers_88_mlp_down_proj/max_abs":2.421875,"train/train/tensor_act_model_layers_47/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_mlp_gate_proj_weight/mean":4.394678398966789e-08,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp/norm":334.58020256501527,"train/train/tensor_act_model_layers_89_mlp_gate_proj/max_abs":3.890625,"train/train/layer__model_layers_75/param/mean":0.0015452916090276424,"train/train/tensor_act_model_layers_11_self_attn_q_proj/max_abs":5.84375,"train/train/tensor_act_model_layers_33_post_attention_layernorm/max_abs":6.6875,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/norm":5.6875,"train/train/tensor_act_model_layers_48_post_attention_layernorm/norm":5792.614868165507,"train/train/layer__model_layers_62/param/mean":0.0016142418157663806,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/max_abs":0.25390625,"train/train/tensor_act_model_layers_63_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91/max_abs":14.5,"train/train/tensor_param_model_layers_24_mlp_gate_proj_weight/std":0.0267333984375,"train/train/tensor_act_model_layers_44_self_attn/std":0.07934697269377686,"train/train/tensor_act_model_layers_26_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_self_attn_k_proj/mean":0.04534912109375,"train/train/tensor_act_model_layers_22_self_attn_k_proj/norm":4569.630616171144,"train/train/tensor_act_model_layers_35_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/norm":0.014262585298348335,"train/train/tensor_act_model_layers_73/std":1.5117269839424428,"train/train/tensor_act_model_layers_11_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_k_proj/std":1.25195776094956,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86/std":2.0156265171852694,"train/train/tensor_grad_model_layers_7_mlp_gate_proj_weight/mean":-1.1938391253352165e-07,"train/train/tensor_act_model_layers_90_mlp_gate_proj/mean":-0.0096435546875,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/std":0.0283203125,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_mlp_gate_proj_weight/norm":0.020182527520227023,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/max_abs":0.11376953125,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/std":2.2576010980280147e-05,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/std":2.5455829744078238e-05,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/norm":0.017372518341619098,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn/max_abs":1.796875,"train/train/tensor_param_model_layers_2_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_77_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/mean":-0.0001697540283203125,"train/train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/mean":0.0004367828369140625,"train/train/layer_model_layers_19/grad/norm":0.039029501317155296,"train/train/tensor_grad_model_layers_39_mlp_gate_proj_weight/mean":-1.41095370054245e-07,"train/train/tensor_act_model_layers_19_post_attention_layernorm/mean":-0.02606201171875,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/norm":0.0006581071455037359,"train/train/tensor_act_model_layers_89_self_attn_k_proj/norm":5450.643703665266,"train/train/tensor_act_model_layers_80_self_attn_o_proj/max_abs":3.578125,"train/train/tensor_act_model_layers_11_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_73_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_76/act/std":0.7500254233760166,"train/train/tensor_act_model_layers_42_mlp_up_proj/norm":2626.672521865871,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_input_layernorm/std":1.0000020380809624,"train/train/tensor_act_model_layers_38_post_attention_layernorm/mean":-0.0123138427734375,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/max_abs":0.228515625,"train/train/tensor_act_model_layers_4_self_attn_q_proj/max_abs":8.8125,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/std":0.03955078125,"train/train/tensor_act_model_layers_92_self_attn_v_proj/norm":2877.5778967145507,"train/train/tensor_act_model_layers_78_mlp_up_proj/std":0.4921876218110645,"train/train/layer_model_layers_68/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/mean":0.00010824203491210938,"train/train/tensor_act_model_layers_78_mlp_gate_proj/norm":4037.116700476601,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_23/std":1.263678553282058,"train/train/tensor_param_model_layers_80_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/norm":0.004839466338878983,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_post_attention_layernorm/max_abs":6.46875,"train/train/tensor_act_model_layers_37_post_attention_layernorm/mean":-0.012542724609375,"train/train/layer__model_layers_13/param/std":0.049483841062430686,"train/train/tensor_param_model_layers_67_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_gate_proj_weight/max_abs":0.142578125,"train/train/layer_model_layers_3/grad/mean":-2.100106979634572e-08,"train/train/layer_model_layers_15/grad/norm":0.036805575734108184,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_post_attention_layernorm/max_abs":5.71875,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/mean":-6.05359673500061e-08,"train/train/tensor_act_model_layers_85_self_attn_k_proj/norm":5888.576224299575,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/norm":0.02522089051683885,"train/train/layer__model_layers_44/param/max_abs":1,"train/train/tensor_act_model_layers_92_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_mlp_gate_proj_weight/norm":5.53125,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/mean":2.7057831175625324e-07,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/mean":1.9115395843982697e-07,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/std":0.057861328125,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/std":3.9178506893012295e-05,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_70_mlp_up_proj/max_abs":2.640625,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/mean":-6.449408829212189e-08,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/max_abs":0.00038909912109375,"train/train/tensor_act_model_layers_80_self_attn/norm":1298.0215039662069,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/std":0.0458984375,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/norm":0.0266165072099722,"train/train/tensor_grad_model_layers_21_mlp_gate_proj_weight/norm":0.014869734958288907,"train/train/tensor_act_model_layers_56_self_attn/max_abs":1.2265625,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/max_abs":0.16015625,"train/train/tensor_act_model_layers_81_self_attn_q_proj/norm":7462.573122455,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/norm":0.00974137155500807,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/mean":0.00028228759765625,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_q_proj/norm":6785.108320016403,"train/learning_rate":0.001,"train/train/layer__model_layers_30/param/norm":20.80233892348406,"train/train/tensor_act_model_layers_51_self_attn_v_proj/max_abs":2.546875,"train/train/tensor_grad_model_layers_56_mlp_gate_proj_weight/mean":7.59027898311615e-08,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_k_proj/norm":4635.116412226574,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/norm":5.25,"train/train/tensor_act_model_layers_92_self_attn_q_proj/mean":0.067626953125,"train/train/tensor_act_model_layers_78_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_41/grad/max_abs":0.00159454345703125,"train/train/tensor_act_model_layers_30_self_attn/norm":417.40750028311476,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/mean":-0.00011444091796875,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/std":0.0001261766293794863,"train/train/tensor_act_model_layers_85_mlp_gate_proj/max_abs":3.171875,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/std":8.302304253726326e-05,"train/train/tensor_act_model_layers_80_input_layernorm/std":1.0000012071097435,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/mean":5.7334545999765396e-08,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/std":0.0255126953125,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/norm":5.75,"train/train/tensor_act_model_layers_73_mlp_gate_proj/norm":3741.0618207506163,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/max_abs":0.0002498626708984375,"train/train/tensor_act_model_layers_81_self_attn_q_proj/max_abs":6.46875,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/norm":0.0013937224990142583,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/mean":2.150190994143486e-07,"train/train/tensor_act_model_layers_23_mlp_up_proj/max_abs":1.953125,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/std":7.537742817741963e-05,"train/train/tensor_act_model_layers_17_self_attn_v_proj/norm":1841.460511194322,"train/train/layer__model_layers_88/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/norm":0.018013955349073685,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/max_abs":0.162109375,"train/train/tensor_act_model_layers_72_self_attn_k_proj/norm":5137.151521416301,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/max_abs":0.0004062652587890625,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/norm":4.65625,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/norm":5.9375,"train/train/tensor_act_model_layers_71_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_81_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_input_layernorm/mean":-0.019561767578125,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/mean":3.6307028494775295e-07,"train/train/tensor_act_model_layers_38_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_mlp_up_proj/std":0.23437503287568456,"train/train/tensor_param_model_layers_65_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_14_mlp_gate_proj_weight/norm":4.65625,"train/train/tensor_act_model_layers_53_self_attn_v_proj/std":0.3789062905557355,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/max_abs":0.0002460479736328125,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_gate_proj/norm":5729.185664267284,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/max_abs":0.000339508056640625,"train/train/tensor_act_model_layers_4_input_layernorm/norm":5792.605834961106,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/mean":-1.3294629752635956e-07,"train/train/tensor_act_model_layers_72_input_layernorm/max_abs":5.5625,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/std":0.038330078125,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/max_abs":0.000194549560546875,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_post_attention_layernorm/mean":0.0047893524169921875,"train/train/tensor_act_model_layers_78_self_attn_o_proj/norm":1086.3418890245548,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/max_abs":0.00054931640625,"train/train/tensor_act_model_layers_93_mlp_down_proj/std":1.1230655801293403,"train/train/layer_model_layers_17/grad/max_abs":0.00141143798828125,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/max_abs":0.150390625,"train/train/tensor_act_model_layers_15_mlp_down_proj/norm":230.43217809229014,"train/train/layer_model_layers_38/act/norm":13897.077622157212,"train/train/tensor_act_model_layers_66_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_gate_proj/max_abs":1.8984375,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/max_abs":0.000400543212890625,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/std":5.1505401792583975e-05,"train/train/tensor_act_model_layers_29_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_post_attention_layernorm/std":1.0000004842876213,"train/train/tensor_act_model_layers_33_self_attn_o_proj/norm":491.82890848715755,"train/train/tensor_act_model_layers_48_mlp_up_proj/max_abs":2.046875,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/std":3.375682611037503e-05,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_73/param/max_abs":1,"train/train/tensor_param_model_layers_47_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_mlp/max_abs":0.51953125,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/norm":0.030314079813738196,"train/train/layer__model_layers_69/param/max_abs":1,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/max_abs":0.111328125,"train/train/tensor_act_model_layers_17_self_attn_q_proj/mean":-0.027313232421875,"train/train/tensor_grad_model_layers_69_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/max_abs":0.2080078125,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/max_abs":0.00020313262939453125,"train/train/tensor_act_model_layers_10_self_attn_q_proj/max_abs":8.625,"train/train/layer_model_layers_82/grad/max_abs":0.00156402587890625,"train/train/tensor_act_model_layers_13_self_attn_v_proj/max_abs":2.15625,"train/train/tensor_act_model_layers_46_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_77/act/mean":-0.00914696284702846,"train/train/tensor_act_model_layers_35_mlp/norm":285.28552406876935,"train/train/tensor_act_model_layers_57_self_attn_q_proj/max_abs":5.65625,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/mean":-3.4477561712265015e-06,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_19/act/std":0.629245484232724,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_k_proj/std":0.8623064513598688,"train/train/tensor_param_model_layers_90_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_59_post_attention_layernorm/norm":5792.61340332294,"train/train/tensor_act_model_layers_46_self_attn_k_proj/mean":0.027099609375,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_input_layernorm/max_abs":6.28125,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_14/param/max_abs":1,"train/train/tensor_act_model_layers_53_self_attn_o_proj/norm":584.5732560041911,"train/train/tensor_act_model_layers_79_self_attn_k_proj/mean":-0.0018286705017089844,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/norm":6.28125,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_mlp_gate_proj/max_abs":2.25,"train/train/layer_model_layers_28/grad/norm":0.0433587601616479,"train/train/layer_model_layers_5/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_input_layernorm/std":1.0000010477373475,"train/train/layer__model_layers_74/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_87/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/norm":0.011352785169333387,"train/train/tensor_act_model_layers_55_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/std":0.052490234375,"train/train/tensor_act_model_layers_81_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/std":5.980844708311155e-05,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/std":0.022216796875,"train/train/tensor_act_model_layers_89_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn_o_proj/norm":1298.823011373343,"train/train/layer_model_layers_70/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/max_abs":0.2470703125,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/max_abs":0.1923828125,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/mean":2.0471634343266487e-07,"train/train/tensor_act_model_layers_24_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/norm":0.0010930954941820966,"train/train/tensor_act_model_layers_47_mlp/mean":0.0002276897430419922,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/norm":8.375,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/mean":-0.000774383544921875,"train/train/tensor_grad_model_layers_84_mlp_gate_proj_weight/mean":-5.876645445823669e-07,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/std":7.285016572568038e-05,"train/train/tensor_act_model_layers_42_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_mlp_gate_proj_weight/std":7.225349753419485e-05,"train/train/layer__model_layers_48/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_25/grad/max_abs":0.0015716552734375,"train/train/tensor_act_model_layers_19_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_42_input_layernorm/std":1.0000002775340888,"train/train/tensor_param_model_layers_87_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/std":0.051513671875,"train/train/layer_model_layers_70/grad/std":7.359418884931794e-05,"train/train/tensor_act_model_layers_13_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_input_layernorm/max_abs":6.6875,"train/train/layer__model_layers_56/param/std":0.05597121103413485,"train/train/tensor_act_model_layers_64_mlp_gate_proj/norm":3368.3571786968864,"train/train/tensor_act_model_layers_37/norm":7219.674982934345,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/norm":0.0012645189207910789,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp_gate_proj/mean":0.002536773681640625,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/std":4.2507528730509846e-05,"train/train/tensor_act_model_layers_12_mlp_down_proj/mean":-0.0001284480094909668,"train/train/tensor_act_model_layers_69_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/std":0.034912109375,"train/train/tensor_act_model_layers_77_self_attn_v_proj/std":0.4658213486449996,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/std":0.033935546875,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/std":9.336939577452843e-05,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/max_abs":0.000553131103515625,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/mean":-0.0002613067626953125,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_v_proj/std":0.4042970059925235,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/norm":0.0010010727975686582,"train/train/tensor_act_model_layers_52_self_attn_q_proj/max_abs":7,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_mlp_gate_proj_weight/mean":-9.860377758741379e-08,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_gate_proj_weight/max_abs":0.1162109375,"train/train/tensor_act_model_layers_35_input_layernorm/max_abs":6.6875,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/std":6.366354343111761e-05,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/std":0.0390625,"train/train/tensor_param_model_layers_20_mlp_gate_proj_weight/norm":4.78125,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_v_proj/mean":-0.00579833984375,"train/train/tensor_act_model_layers_85_mlp_up_proj/norm":4532.308760359695,"train/train/layer__model_layers_47/param/norm":21.683745397258402,"train/train/tensor_act_model_layers_65_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/mean":1.0069925338029861e-07,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_38_mlp_gate_proj_weight/std":0.0302734375,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_post_attention_layernorm/norm":5792.6099853526875,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/max_abs":0.20703125,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/norm":4.71875,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/std":0.02392578125,"train/train/tensor_act_model_layers_43_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/norm":6.0625,"train/train/layer__model_layers_65/param/max_abs":1,"train/train/tensor_act_model_layers_73_self_attn_q_proj/norm":5778.969468267322,"train/train/layer_model_layers_4/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/norm":0.030649981333180854,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/mean":-1.436471939086914e-05,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/mean":1.4039687812328339e-07,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/mean":-7.236376404762268e-07,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/mean":-4.6798959374427795e-07,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/mean":-2.891756594181061e-07,"train/train/tensor_act_model_layers_36_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/norm":5.75,"train/train/tensor_act_model_layers_53_mlp_up_proj/max_abs":2.109375,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_k_proj/max_abs":4.8125,"train/train/tensor_act_model_layers_82_mlp_up_proj/mean":-0.0013561248779296875,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/norm":0.0010527942473242424,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/mean":0.062255859375,"train/train/layer__model_layers_76/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn_o_proj/norm":550.3042435967333,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/std":0.0294189453125,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/mean":0.00012493133544921875,"train/train/tensor_act_model_layers_76_self_attn_v_proj/std":0.42382813681105846,"train/train/tensor_act_model_layers_61_post_attention_layernorm/std":1.0000005794992743,"train/train/tensor_param_model_layers_71_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_gate_proj_weight/std":0.03662109375,"train/train/tensor_act_model_layers_72_post_attention_layernorm/norm":5792.609863284317,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/mean":-2.2977590560913086e-05,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/max_abs":0.00060272216796875,"train/train/tensor_param_model_layers_3_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/norm":0.022674786449760027,"train/train/tensor_act_model_layers_25_mlp_down_proj/mean":0.00121307373046875,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp_down_proj/mean":0.0004096031188964844,"train/train/tensor_param_model_layers_32_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_61/grad/max_abs":0.0024871826171875,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/norm":9.8125,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/max_abs":0.000263214111328125,"train/train/tensor_act_model_layers_81_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_q_proj/mean":0.0005991458892822266,"train/train/tensor_param_model_layers_64_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_48/mean":-0.0138702392578125,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp_gate_proj/max_abs":1.625,"train/train/tensor_act_model_layers_66_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_post_attention_layernorm/std":1.0000005166450394,"train/train/tensor_act_model_layers_88_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/std":0.00023998307473017402,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_mlp_down_proj/norm":395.6589739827596,"train/train/tensor_act_model_layers_57_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/norm":0.03085802534025065,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/std":7.619167731326906e-05,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/mean":-0.0004673004150390625,"train/train/tensor_act_model_layers_43_self_attn/std":0.11133511007597366,"train/train/tensor_act_model_layers_16/mean":-0.032806396484375,"train/train/tensor_act_model_layers_14_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp_gate_proj/std":0.4062502974152927,"train/train/tensor_act_model_layers_17_mlp_up_proj/std":0.2519531818323293,"train/train/tensor_act_model_layers_86_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_norm_weight/max_abs":1,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/std":9.976633867149988e-05,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_v_proj/std":0.39355590365489557,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/norm":4.34375,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/std":0.03515625,"train/train/tensor_act_model_layers_56_self_attn_k_proj/norm":5261.375697991503,"train/train/tensor_grad_model_layers_16_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn/mean":-0.0010423660278320312,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/std":5.3481439835106696e-05,"train/train/tensor_act_model_layers_17_self_attn_q_proj/std":1.0703130790785456,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/norm":5.375,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/max_abs":0.0003414154052734375,"train/train/tensor_act_model_layers_92/std":2.824229634792937,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn/mean":0.0005002021789550781,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_mlp_gate_proj_weight/max_abs":0.00079345703125,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/max_abs":0.0002899169921875,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/mean":1.5079975128173828e-05,"train/train/tensor_act_model_layers_12_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_5_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_v_proj/norm":1707.759666788574,"train/train/tensor_act_model_layers_81_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp_down_proj/norm":836.1896726311006,"train/train/tensor_act_model_layers_90_self_attn/norm":1389.7820880256709,"train/train/layer_model_layers_10/grad/mean":8.740447833552933e-08,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/mean":3.3690594136714935e-07,"train/train/tensor_act_model_layers_88_self_attn/max_abs":2.015625,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/max_abs":0.107421875,"train/train/tensor_act_model_layers_80_mlp_up_proj/max_abs":2.890625,"train/train/tensor_act_model_layers_1_self_attn/std":0.01617914217245331,"train/train/tensor_act_model_layers_63_self_attn_o_proj/max_abs":1.296875,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/mean":4.9985828809440136e-08,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/mean":9.506940841674805e-06,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/max_abs":0.0001010894775390625,"train/train/tensor_act_model_layers_5_self_attn_k_proj/norm":6068.512927079498,"train/train/tensor_act_model_layers_79_self_attn_k_proj/std":0.8652370535903438,"train/train/layer_model_layers_26/grad/mean":-1.38753987325707e-10,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn_v_proj/max_abs":2.546875,"train/train/tensor_act_model_layers_29_mlp_down_proj/max_abs":0.357421875,"train/train/tensor_act_model_layers_80_self_attn_v_proj/std":0.4990245624881045,"train/train/tensor_act_model_layers_58_mlp/mean":0.0010890960693359375,"train/train/tensor_act_model_layers_10_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_self_attn_v_proj/max_abs":2.265625,"train/train/tensor_act_model_layers_67_self_attn_q_proj/max_abs":5.59375,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_45/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_gate_proj/std":0.25585947654968716,"train/train/tensor_act_model_layers_58/max_abs":9.8125,"train/train/tensor_act_model_layers_59_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_mlp_gate_proj_weight/std":4.1744059441910445e-05,"train/train/tensor_act_model_layers_79_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_75/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_down_proj/norm":330.3591014355224,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_28_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_24/grad/std":3.618332281127754e-05,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/max_abs":0.00055694580078125,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/norm":0.03239949384576543,"train/train/tensor_act_model_layers_90_mlp_up_proj/std":0.6640626732561073,"train/train/tensor_act_model_layers_90_self_attn_v_proj/mean":0.0022058486938476562,"train/train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_66/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/std":0.0294189453125,"train/train/layer__model_layers_57/param/max_abs":1,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_58/act/std":0.6584507509751905,"train/train/tensor_act_model_layers_61_self_attn_o_proj/max_abs":2.96875,"train/train/tensor_act_model_layers_67_mlp/norm":699.4997944017726,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_43_input_layernorm/max_abs":6.53125,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/norm":0.009060432017265035,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_v_proj/mean":0.00284576416015625,"train/train/tensor_act_model_layers_0_self_attn_o_proj/std":0.03344728437380979,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/mean":-9.298324584960938e-05,"train/train/tensor_act_model_layers_54_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_mlp_up_proj/max_abs":1.8828125,"train/train/layer_model_layers_28/grad/std":5.353039173229542e-05,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/std":0.03857421875,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/max_abs":0.2236328125,"train/train/tensor_act_model_layers_88_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/mean":1.2782402336597443e-07,"train/train/tensor_act_model_layers_50_self_attn_v_proj/std":0.36865337020996763,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/norm":0.028567061345932326,"train/train/tensor_act_model_layers_75_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_73_self_attn_v_proj/mean":0.0016269683837890625,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/mean":-2.2504536900669336e-07,"train/train/tensor_act_model_layers_28_mlp_up_proj/std":0.26171886125827315,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_input_layernorm/std":1.0000015734682517,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_input_layernorm/std":1.0000002724118158,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/std":8.64529947618649e-05,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/norm":0.020825692603556586,"train/train/tensor_act_model_layers_11_self_attn/max_abs":1.0703125,"train/train/tensor_act_model_layers_71_self_attn_q_proj/norm":5478.41841685092,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/std":0.037353515625,"train/train/tensor_act_model_layers_20_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/std":5.646290038308806e-05,"train/train/layer_model_layers_50/grad/max_abs":0.0016937255859375,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/mean":3.577442839741707e-07,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/mean":-8.573988452553749e-08,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/norm":0.015044423175912284,"train/train/tensor_act_model_layers_75_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_57_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_29/act/max_abs":9,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/mean":1.201988197863102e-07,"train/train/tensor_act_model_layers_27_self_attn/std":0.06494519366741308,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/max_abs":0.1318359375,"train/train/tensor_act_model_layers_47_mlp_up_proj/mean":-0.006256103515625,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/mean":-0.000118255615234375,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/mean":-1.0710209608078003e-08,"train/train/tensor_act_model_layers_53_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp_gate_proj/mean":0.00251007080078125,"train/train/tensor_act_model_layers_43_mlp/max_abs":0.427734375,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_post_attention_layernorm/std":1.0000001396983764,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_84_mlp/norm":1365.1432541772751,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/mean":-2.1560117602348328e-07,"train/train/tensor_act_model_layers_93_mlp/norm":6506.932066345242,"train/train/tensor_act_model_layers_45_mlp_up_proj/mean":-0.00496673583984375,"train/train/layer__model_layers_13/param/max_abs":1,"train/train/tensor_grad_model_layers_17_mlp_gate_proj_weight/norm":0.015903201596148133,"train/train/tensor_act_model_layers_28_self_attn_o_proj/norm":559.808692063593,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/max_abs":0.11474609375,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/norm":0.0032016470308021696,"train/train/tensor_act_model_layers_58_input_layernorm/std":1.000000668574334,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_41/act/std":0.6315354724437893,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/mean":0.000274658203125,"train/train/tensor_act_model_layers_72_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_46/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_post_attention_layernorm/std":1.0000006938350774,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/norm":5.03125,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/std":0.037109375,"train/train/tensor_act_model_layers_40_mlp_gate_proj/mean":0.00746917724609375,"train/train/layer_model_layers_91/act/norm":21990.834476305918,"train/train/tensor_act_model_layers_55/mean":-0.002933502197265625,"train/train/tensor_param_model_layers_61_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/max_abs":0.0015106201171875,"train/train/tensor_act_model_layers_38_mlp_gate_proj/max_abs":1.9921875,"train/train/tensor_act_model_layers_93_mlp/std":1.1230655801293403,"train/train/tensor_act_model_layers_92_self_attn_q_proj/std":1.228520986184459,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/max_abs":0.15234375,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/mean":-3.674067556858063e-07,"train/train/tensor_act_model_layers_82_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/norm":1210.7324515503085,"train/train/tensor_param_model_layers_75_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_40/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_mlp_gate_proj/mean":-0.005096435546875,"train/train/tensor_act_model_layers_32/mean":-0.01800537109375,"train/train/tensor_grad_model_layers_60_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_mlp_gate_proj_weight/norm":6.65625,"train/train/tensor_act_model_layers_48_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/max_abs":0.000720977783203125,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/max_abs":0.000640869140625,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/max_abs":0.0004444122314453125,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/std":0.039306640625,"train/train/tensor_act_model_layers_77_mlp_down_proj/mean":0.0015316009521484375,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/max_abs":0.15625,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn/mean":0.0011234283447265625,"train/train/tensor_act_model_layers_7_self_attn_o_proj/std":0.04614781258040305,"train/train/tensor_act_model_layers_48_post_attention_layernorm/mean":-0.01165771484375,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_gate_proj/max_abs":1.7109375,"train/train/tensor_act_model_layers_29_mlp_gate_proj/norm":2233.7203289520335,"train/train/layer_model_layers_31/grad/norm":0.036906567895950454,"train/train/layer__model_layers_2/param/std":0.04738348804518912,"train/train/tensor_act_model_layers_68_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_embed_tokens_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_k_proj/std":0.89062532050562,"train/train/tensor_act_model_layers_63_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/norm":0.008217180222730937,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/std":9.387893991344683e-06,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/norm":6,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/norm":0.0011199499011981975,"train/train/tensor_act_model_layers_87_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/max_abs":0.0004596710205078125,"train/train/tensor_act_model_layers_5_mlp_gate_proj/mean":7.301568984985352e-05,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/max_abs":0.000286102294921875,"train/train/tensor_act_model_layers_74_self_attn_k_proj/max_abs":5.65625,"train/train/tensor_act_model_layers_86_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/max_abs":0.00128936767578125,"train/train/tensor_param_model_layers_17_mlp_gate_proj_weight/max_abs":0.130859375,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/norm":6.59375,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_66/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/mean":-6.084028427721933e-07,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_49_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_19_post_attention_layernorm/norm":5792.603881840788,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/norm":0.01406112107827797,"train/train/tensor_act_model_layers_25_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/mean":1.5683472156524658e-06,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/norm":6.15625,"train/train/tensor_act_model_layers_71_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_61/param/std":0.05900021267299283,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/norm":4.75,"train/train/layer__model_layers_7/param/mean":0.001541125792980938,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_input_layernorm/norm":5792.599853521146,"train/train/tensor_act_model_layers_38_post_attention_layernorm/norm":5792.600585940197,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/norm":0.026922800447805047,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/std":5.736938224627649e-05,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/mean":0.00018596649169921875,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/mean":2.427259460091591e-07,"train/train/tensor_act_model_layers_36_input_layernorm/max_abs":6.71875,"train/train/tensor_grad_model_layers_45_mlp_gate_proj_weight/std":4.955338659727734e-05,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/max_abs":0.001129150390625,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/norm":0.004131132211535354,"train/train/tensor_param_model_layers_70_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_mlp_gate_proj_weight/max_abs":0.1298828125,"train/train/tensor_act_model_layers_44_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp_gate_proj/norm":2034.2469201541019,"train/train/tensor_act_model_layers_81_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/std":5.879891800135672e-05,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/std":0.0262451171875,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/max_abs":0.000423431396484375,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/norm":3.828125,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/mean":7.343292236328125e-05,"train/train/tensor_act_model_layers_23_self_attn_k_proj/norm":4211.992201681276,"train/train/tensor_act_model_layers_66_input_layernorm/std":1.0000004287793662,"train/train/tensor_act_model_layers_5_mlp_down_proj/mean":-0.0003490447998046875,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/std":0.045654296875,"train/train/tensor_act_model_layers_93/std":3.378915067996112,"train/train/tensor_param_model_layers_34_mlp_gate_proj_weight/norm":5.25,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/mean":4.8317015171051025e-06,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/norm":0.000578212981842961,"train/train/tensor_act_model_layers_22_mlp/norm":210.19898898132539,"train/train/tensor_act_model_layers_73_mlp_up_proj/std":0.4599621414236645,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/norm":4.375,"train/train/tensor_act_model_layers_83_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/norm":0.00112844780379453,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/mean":0.00020694732666015625,"train/train/tensor_act_model_layers_68_self_attn_o_proj/max_abs":4.75,"train/train/tensor_act_model_layers_39_self_attn_v_proj/mean":0.0005974769592285156,"train/train/layer_model_layers_19/grad/std":4.8162687976408336e-05,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/norm":0.003878089608492805,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/mean":-8.535385131835938e-05,"train/train/tensor_param_model_layers_83_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/mean":-3.9814040064811707e-08,"train/train/tensor_act_model_layers_61_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_84/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/norm":5.8125,"train/train/tensor_act_model_layers_66_mlp_up_proj/std":0.42773465669309146,"train/train/tensor_act_model_layers_29_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/mean":-4.363246262073517e-07,"train/train/tensor_act_model_layers_41_mlp_down_proj/norm":333.138692138353,"train/train/layer__model_layers_17/param/mean":0.0015880168879087928,"train/train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_mlp_gate_proj_weight/mean":-9.546056389808655e-09,"train/train/layer_model_layers_42/grad/max_abs":0.0013885498046875,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/norm":3.265625,"train/train/tensor_act_model_layers_37/max_abs":9.3125,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/std":1.8689064252277747e-05,"train/train/tensor_act_model_layers_56_post_attention_layernorm/norm":5792.60827636856,"train/train/tensor_act_model_layers_65_mlp/mean":0.0014019012451171875,"train/train/tensor_act_model_layers_20_input_layernorm/max_abs":6.3125,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/max_abs":0.000499725341796875,"train/train/tensor_act_model_layers_60_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/mean":-4.8814399633556604e-08,"train/train/total_time_seconds":2289.6543385237455,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/norm":5.375,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/norm":3.796875,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/mean":6.241723895072937e-06,"train/train/layer__model_layers_19/param/std":0.04986226386403908,"train/train/layer_model_layers_69/grad/max_abs":0.00118255615234375,"train/train/tensor_act_model_layers_86_input_layernorm/norm":5792.603759766556,"train/train/tensor_act_model_layers_21_mlp_gate_proj/std":0.250000150659912,"train/train/tensor_act_model_layers_25_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/norm":0.0026702335897061872,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/norm":0.01836937672505402,"train/train/layer__model_layers_37/param/std":0.05351372787902525,"train/train/tensor_act_model_layers_9_input_layernorm/norm":5792.609619143408,"train/train/tensor_param_model_layers_48_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_13_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_o_proj/max_abs":1.5703125,"train/train/layer_model_layers_58/grad/max_abs":0.00127410888671875,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/std":0.045166015625,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/max_abs":0.00055694580078125,"train/train/tensor_act_model_layers_82_self_attn_k_proj/std":0.9521503839717433,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_mlp_gate_proj_weight/mean":-0.0003185272216796875,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/max_abs":0.123046875,"train/train/layer__model_layers_38/param/max_abs":1,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/norm":0.018043090621691914,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/max_abs":0.0002918243408203125,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/norm":0.004985702793702286,"train/train/tensor_act_model_layers_54_self_attn_o_proj/mean":0.0009241104125976562,"train/train/layer__model_layers_49/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/mean":-0.0003662109375,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/norm":5.40625,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_v_proj/mean":0.0124969482421875,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/std":0.0238037109375,"train/train/tensor_param_model_layers_68_mlp_gate_proj_weight/std":0.038818359375,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/std":2.656147797912434e-05,"train/train/layer_model_layers_30/grad/std":4.486870582665126e-05,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/norm":5,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_input_layernorm/mean":-0.0123138427734375,"train/train/tensor_act_model_layers_67_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/norm":0.022510556684979716,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/max_abs":0.00174713134765625,"train/train/tensor_act_model_layers_92_self_attn_v_proj/std":0.4970719190123777,"train/train/tensor_act_model_layers_66_self_attn_q_proj/norm":5481.706810711559,"train/train/tensor_act_model_layers_57_post_attention_layernorm/max_abs":5.96875,"train/train/layer_model_layers_88/grad/mean":-2.7211763650504362e-08,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp_down_proj/norm":781.7845838840727,"train/train/layer_model_layers_63/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_mlp/max_abs":1.4765625,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/std":9.575156792905701e-05,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/mean":6.162095814943314e-06,"train/train/tensor_act_model_layers_47_self_attn_k_proj/norm":4205.622996581418,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/norm":3.0625,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/max_abs":0.0021820068359375,"train/train/tensor_grad_model_layers_69_mlp_gate_proj_weight/norm":0.025577007802940963,"train/train/tensor_param_model_layers_3_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/mean":3.366731107234955e-07,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/std":6.691689337210015e-05,"train/train/layer_model_layers_66/act/mean":0.004448115825653076,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/norm":0.02290990180535255,"train/train/tensor_act_model_layers_52_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26/norm":7241.08784177268,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_33/param/mean":0.0015500941998129144,"train/train/tensor_act_model_layers_65/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_mlp_gate_proj_weight/mean":-2.191518433392048e-08,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_80_self_attn_q_proj/std":1.201176706940151,"train/train/tensor_param_model_layers_53_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_2_mlp/mean":-0.0043487548828125,"train/grad_norm":0.80078125,"train/train/tensor_act_model_layers_38_self_attn_v_proj/mean":-0.00528717041015625,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/mean":1.4959368854761124e-08,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_mlp_gate_proj_weight/norm":4.4375,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_v_proj/norm":2086.975498371578,"train/train/layer_model_layers_28/grad/mean":-4.7857775796036265e-08,"train/train/tensor_grad_model_layers_47_mlp_gate_proj_weight/std":5.169993451782903e-05,"train/train/tensor_act_model_layers_54_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_k_proj/max_abs":5.1875,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_q_proj/mean":0.0430908203125,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/std":0.00011659593108312209,"train/train/tensor_act_model_layers_4_mlp/max_abs":1.28125,"train/train/layer__model_layers_64/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/std":1.9039357865899314e-05,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/mean":-9.107589721679688e-05,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/norm":3.59375,"train/train/tensor_act_model_layers_64_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/max_abs":0.00013065338134765625,"train/train/tensor_grad_model_layers_13_mlp_gate_proj_weight/max_abs":0.000644683837890625,"train/train/layer_model_layers_53/grad/mean":-1.3250642559271706e-09,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/norm":0.017417812472470283,"train/train/tensor_act_model_layers_78_mlp/std":0.1784673513232102,"train/train/tensor_act_model_layers_50_self_attn_q_proj/std":0.8779319512801639,"train/train/tensor_act_model_layers_34_mlp/std":0.047058246208170944,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_mlp_up_proj/norm":3045.4430866700427,"train/train/tensor_act_model_layers_40_mlp/std":0.05706797547549185,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/norm":0.019176115491964883,"train/train/layer__model_layers_28/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp_gate_proj/mean":0.01031494140625,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/norm":7.25,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_mlp_gate_proj_weight/norm":0.015003882789742446,"train/train/tensor_param_model_layers_80_input_layernorm_weight/std":0,"train/train/layer__model_layers_57/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/std":7.786114436452014e-05,"train/train/layer_model_layers_60/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/mean":5.925539880990982e-08,"train/train/tensor_act_model_layers_22_self_attn_k_proj/max_abs":4.375,"train/train/tensor_grad_model_layers_13_mlp_gate_proj_weight/mean":5.0175003707408905e-08,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/std":5.563394318131255e-05,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/mean":1.7680213204585016e-07,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/std":1.764852241090396e-05,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/std":0.037109375,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/max_abs":0.000560760498046875,"train/train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp/std":0.09191922542990939,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/mean":1.1775409802794456e-07,"train/train/layer_model_layers_59/act/std":0.6741556551779718,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_k_proj/max_abs":5,"train/train/tensor_grad_model_layers_75_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn/max_abs":1.3125,"train/train/tensor_act_model_layers_41_mlp_up_proj/mean":-0.002735137939453125,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/norm":7.34375,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/norm":0.0036717288515752374,"train/train/tensor_act_model_layers_6_self_attn_q_proj/std":1.1171875983685002,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn_q_proj/mean":0.00502777099609375,"train/train/tensor_act_model_layers_16_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp/norm":799.4549702821303,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/std":4.055480275211725e-05,"train/train/tensor_act_model_layers_76_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71/norm":8583.764906249235,"train/train/tensor_act_model_layers_20_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_mlp_gate_proj_weight/mean":-5.529727786779404e-08,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/max_abs":0.130859375,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_q_proj/max_abs":5.46875,"train/train/tensor_act_model_layers_24_self_attn_v_proj/norm":1506.0832102733805,"train/train/tensor_act_model_layers_58_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/std":2.8572779319951135e-05,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn_v_proj/std":0.318359584598933,"train/train/tensor_act_model_layers_55_self_attn_k_proj/norm":4444.271534796437,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/mean":-0.00067138671875,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_o_proj/max_abs":1.3125,"train/train/tensor_param_model_layers_60_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59/std":1.3125008947134622,"train/train/tensor_act_model_layers_41_mlp_gate_proj/mean":0.00107574462890625,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88/mean":0.014404296875,"train/train/tensor_act_model_layers_37_self_attn_k_proj/norm":4746.997991763627,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn_k_proj/std":1.0605525334933357,"train/train/tensor_act_model_layers_47_self_attn_k_proj/std":0.7265625480682604,"train/train/tensor_act_model_layers_6_mlp/norm":407.2217637710753,"train/train/tensor_act_model_layers_82_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/norm":7.28125,"train/train/tensor_act_model_layers_68_self_attn_k_proj/std":0.8691451169015773,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_27/grad/norm":0.03689822085365616,"train/train/layer__model_layers_11/param/norm":20.004937134758134,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/norm":4.90625,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/norm":0.03663207843066998,"train/train/layer_model_layers_16/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn/mean":-0.0005030632019042969,"train/train/tensor_act_model_layers_47_input_layernorm/std":1.0000002989545016,"train/train/tensor_act_model_layers_81/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/max_abs":0.0003662109375,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_q_proj/std":0.8359375356513755,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/max_abs":0.0002956390380859375,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/mean":-1.1362135410308838e-07,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/std":0.057373046875,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/std":0.024658203125,"train/train/tensor_act_model_layers_56_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_19/act/mean":-0.0085458585194179,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/mean":1.5337718650698662e-07,"train/train/tensor_param_model_layers_32_mlp_gate_proj_weight/max_abs":0.1376953125,"train/train/layer__model_layers_86/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_mlp/max_abs":0.458984375,"train/train/tensor_act_model_layers_21_self_attn_o_proj/mean":0.0013284683227539062,"train/train/layer__model_layers_38/param/std":0.05259339279879915,"train/train/layer_model_layers_47/grad/norm":0.043264161287463465,"train/train/tensor_grad_model_layers_27_mlp_gate_proj_weight/norm":0.01517352888951647,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/max_abs":0.17578125,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/max_abs":0.0003509521484375,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/mean":-9.107589721679688e-05,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_35_mlp_gate_proj_weight/mean":-0.00013828277587890625,"train/train/tensor_act_model_layers_0_self_attn_k_proj/max_abs":2.78125,"train/train/tensor_act_model_layers_39_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_input_layernorm/mean":-0.0121307373046875,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/norm":0.006344005561951924,"train/train/tensor_act_model_layers_59_self_attn_q_proj/std":1.0156254163154337,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/norm":0.00529200675358363,"train/train/tensor_act_model_layers_59_input_layernorm/mean":0.0010099411010742188,"train/train/tensor_act_model_layers_67_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_mlp_gate_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/std":0.0003179513292821848,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/norm":0.016408697581506426,"train/train/tensor_act_model_layers_21_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_75/grad/std":7.806085802057836e-05,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_gate_proj_weight/mean":0.00031280517578125,"train/train/tensor_act_model_layers_76_self_attn_q_proj/std":1.078125170499505,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/std":0.031005859375,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/norm":0.0011553306783069416,"train/train/tensor_act_model_layers_15/norm":7407.052049637725,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/std":0.0341796875,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/norm":0.0007260637678601231,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/max_abs":0.2490234375,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp_gate_proj/norm":1942.6595852609507,"train/train/tensor_act_model_layers_76_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_13_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_post_attention_layernorm/mean":-0.032470703125,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/std":9.061687661287603e-05,"train/train/tensor_act_model_layers_46_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/max_abs":0.000499725341796875,"train/train/tensor_act_model_layers_3_mlp_down_proj/std":0.11780109401055074,"train/train/tensor_act_model_layers_66_self_attn_v_proj/mean":-0.002349853515625,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/mean":-0.000698089599609375,"train/train/tensor_act_model_layers_3_self_attn_q_proj/max_abs":7.59375,"train/train/tensor_act_model_layers_3_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_post_attention_layernorm/max_abs":5.75,"train/train/tensor_act_model_layers_43_post_attention_layernorm/std":1.0000002806773016,"train/train/tensor_act_model_layers_76_self_attn_q_proj/norm":6257.377970126451,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_41/param/max_abs":1,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/mean":1.58033799380064e-08,"train/train/tensor_act_model_layers_23_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/mean":2.9727816581726074e-06,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_gate_proj_weight/mean":0.00023555755615234375,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_post_attention_layernorm/std":1.0000008208440971,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/norm":5.6875,"train/train/tensor_act_model_layers_17_post_attention_layernorm/norm":5792.610229492613,"train/train/tensor_act_model_layers_6_mlp_up_proj/max_abs":2.265625,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/std":0.038818359375,"train/train/tensor_param_model_layers_22_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/norm":5.03125,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/max_abs":0.000621795654296875,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/max_abs":0.16015625,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_o_proj/max_abs":2.296875,"train/train/layer_model_layers_60/grad/mean":1.2853559158893532e-07,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/mean":-4.267692565917969e-05,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/std":3.929098471708683e-05,"train/train/tensor_act_model_layers_28/std":1.24805309252317,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/std":7.317175413584697e-05,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_56/act/mean":-0.0010351496083395822,"train/train/tensor_act_model_layers_80_mlp_gate_proj/norm":4149.35435969283,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/std":4.800441099221916e-05,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_41/grad/std":4.999999444132558e-05,"train/train/tensor_act_model_layers_36_self_attn_o_proj/mean":-0.0008983612060546875,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/max_abs":0.00061798095703125,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/mean":2.9921531677246094e-05,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_v_proj/max_abs":2.609375,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_69_mlp_gate_proj/max_abs":2.5625,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/norm":0.0152514936172998,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/max_abs":0.1337890625,"train/train/tensor_act_model_layers_68_post_attention_layernorm/max_abs":5.5625,"train/train/tensor_act_model_layers_15_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn_q_proj/std":1.1718753306070497,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/std":6.607530796096377e-05,"train/train/layer_model_layers_38/grad/std":5.127728282725902e-05,"train/train/tensor_act_model_layers_69_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_20/grad/norm":0.0309034307926995,"train/train/tensor_act_model_layers_77_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/norm":3.1875,"train/train/tensor_act_model_layers_17/std":1.2734511567038775,"train/train/tensor_act_model_layers_44_mlp/std":0.061767590520699765,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/std":7.349446069965124e-05,"train/train/tensor_act_model_layers_12_self_attn_o_proj/std":0.07629508940315724,"train/train/tensor_act_model_layers_63_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/global/param/max_abs":1,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/std":0.0283203125,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_q_proj/norm":5410.218121566454,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/mean":1.2954697012901306e-06,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/mean":-5.37186861038208e-06,"train/train/tensor_grad_model_layers_50_mlp_gate_proj_weight/std":5.391581010646591e-05,"train/train/tensor_grad_model_layers_90_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60/norm":7621.705692696314,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/max_abs":0.00011682510375976562,"train/train/tensor_act_model_layers_88_self_attn_k_proj/norm":5862.691928735871,"train/train/tensor_act_model_layers_41_mlp/std":0.05749526558990268,"train/train/layer__model_layers_75/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_6/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_v_proj/std":0.2763691768768407,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_50_mlp_down_proj/mean":0.00017714500427246094,"train/train/tensor_act_model_layers_53_mlp/std":0.07470704115367144,"train/train/layer_model_layers_75/act/norm":16019.93314372907,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/mean":1.228763721883297e-07,"train/train/tensor_act_model_layers_53_self_attn_q_proj/std":0.9052757834894838,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/norm":5.65625,"train/train/tensor_grad_model_layers_40_mlp_gate_proj_weight/std":4.901577109825334e-05,"train/train/tensor_act_model_layers_34_self_attn_v_proj/norm":1992.6723490030652,"train/train/tensor_act_model_layers_82/norm":10657.901021101883,"train/train/tensor_act_model_layers_1_self_attn_k_proj/mean":0.0592041015625,"train/train/tensor_grad_model_layers_57_mlp_gate_proj_weight/mean":-7.118796929717064e-08,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/max_abs":0.19921875,"train/train/layer_model_layers_41/grad/mean":3.8440666174553856e-08,"train/train/tensor_act_model_layers_29_self_attn_v_proj/norm":2061.8209468122463,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/mean":0.00017070770263671875,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/mean":4.9173831939697266e-06,"train/train/tensor_act_model_layers_10_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_up_proj/norm":2106.3479324064383,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/max_abs":0.00225830078125,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/norm":3.671875,"train/train/tensor_grad_model_layers_65_mlp_gate_proj_weight/mean":1.6135163605213165e-07,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/mean":-0.00055694580078125,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/std":9.111966752161247e-05,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_mlp_gate_proj_weight/mean":-8.841743692755699e-08,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/max_abs":0.0002231597900390625,"train/train/tensor_act_model_layers_70_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_input_layernorm/std":1.0000016357503492,"train/train/tensor_act_model_layers_61_mlp_down_proj/max_abs":0.7890625,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23/max_abs":8.5,"train/train/layer_model_layers_22/grad/norm":0.03283303266225754,"train/train/tensor_act_model_layers_61_post_attention_layernorm/norm":5792.61022949465,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/norm":2.46875,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/norm":7.59375,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/mean":2.8265640139579773e-07,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_k_proj/norm":4357.763253088328,"train/train/tensor_act_model_layers_70_input_layernorm/std":1.0000017140977295,"train/train/layer_model_layers_9/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_q_proj/mean":0.04412841796875,"train/train/tensor_param_model_layers_33_input_layernorm_weight/max_abs":1,"train/train/layer__model_layers_54/param/max_abs":1,"train/train/tensor_param_model_layers_79_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_54_input_layernorm/max_abs":5.90625,"train/train/tensor_act_model_layers_11_self_attn/norm":303.4659733248781,"train/train/layer_model_layers_91/grad/mean":1.5172857483166055e-07,"train/train/layer_model_layers_14/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_v_proj/max_abs":3.109375,"train/train/layer_model_layers_85/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/max_abs":0.0007476806640625,"train/train/tensor_act_model_layers_69_mlp_up_proj/norm":3648.519017202185,"train/train/tensor_act_model_layers_29_self_attn_q_proj/mean":-0.02618408203125,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/mean":-2.360902726650238e-07,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/norm":4.8125,"train/train/layer_model_layers_68/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_gate_proj/max_abs":2.015625,"train/train/tensor_act_model_layers_59_post_attention_layernorm/std":1.0000006292987402,"train/train/global/param/mean":0.0015258107243339637,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/mean":-6.50063157081604e-07,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/std":0.00011872968083401401,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_26/param/mean":0.0016333688625865348,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/mean":1.9621802493929863e-07,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/mean":1.0371208190917969e-05,"train/train/tensor_param_model_layers_82_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_gate_proj/std":0.3320312649669013,"train/train/tensor_act_model_layers_11_self_attn_k_proj/max_abs":5.3125,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_52/grad/mean":1.981557679599421e-09,"train/train/layer_model_layers_79/act/max_abs":11.5,"train/train/tensor_act_model_layers_38_self_attn_q_proj/norm":5579.738539542691,"train/train/tensor_grad_model_layers_85_mlp_gate_proj_weight/mean":-2.6979250833392143e-07,"train/train/layer_model_layers_25/grad/std":4.616275831873428e-05,"train/train/tensor_act_model_layers_22_mlp_up_proj/std":0.2478031529798284,"train/train/tensor_act_model_layers_9_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/mean":-0.00018787384033203125,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/std":3.8052727792802343e-05,"train/train/tensor_act_model_layers_16_self_attn_o_proj/std":0.08386292271840698,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_60/std":1.3164129275051257,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/std":0.047119140625,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/std":3.203728612770012e-05,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_post_attention_layernorm/norm":5792.604736331269,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/mean":0.00012683868408203125,"train/train/layer__model_layers_92/param/std":0.06616883851688757,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_mlp/norm":469.1437218968507,"train/train/tensor_act_model_layers_3_input_layernorm/mean":-0.025848388671875,"train/train/tensor_act_model_layers_62_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/std":0.028076171875,"train/train/tensor_act_model_layers_2_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer__model_layers_82/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_69_input_layernorm/std":1.000001829581524,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/mean":-7.62939453125e-05,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/mean":2.878950908780098e-07,"train/train/tensor_act_model_layers_3_mlp/std":0.11780109401055074,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/max_abs":0.00141143798828125,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_gate_proj_weight/mean":-1.755543053150177e-07,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_gate_proj/norm":2302.2890566490287,"train/train/tensor_act_model_layers_80_self_attn_v_proj/max_abs":4.15625,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/norm":3.296875,"train/train/tensor_act_model_layers_86_mlp_down_proj/std":0.27002096981450907,"train/train/tensor_act_model_layers_9_mlp_gate_proj/norm":1896.7407145074678,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/max_abs":0.00040435791015625,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/max_abs":0.1357421875,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/std":5.385057067756889e-05,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/std":0.03955078125,"train/train/tensor_act_model_layers_31_mlp_gate_proj/std":0.2792969289450639,"train/train/tensor_act_model_layers_58_post_attention_layernorm/max_abs":5.84375,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/norm":0.03663045127842689,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/norm":6.8125,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_k_proj/max_abs":6,"train/train/tensor_act_model_layers_69_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_post_attention_layernorm/std":1.0000004619358904,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/norm":6.65625,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_input_layernorm/mean":-0.01123046875,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn_k_proj/norm":5300.867333850991,"train/train/tensor_act_model_layers_75_mlp_down_proj/mean":-0.0020122528076171875,"train/train/tensor_act_model_layers_56_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_input_layernorm/mean":0.00641632080078125,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/std":6.537387407611619e-05,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/std":3.193888380813204e-05,"train/train/tensor_act_model_layers_13_mlp/max_abs":0.89453125,"train/train/tensor_act_model_layers_7_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/max_abs":0.0018157958984375,"train/train/tensor_param_model_layers_88_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/max_abs":0.1240234375,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15/mean":-0.03302001953125,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/std":4.360312757988436e-05,"train/train/tensor_act_model_layers_32_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/max_abs":0.0003681182861328125,"train/train/layer_model_layers_29/grad/frac_near_user_limit":0,"train/train/layer_model_layers_26/act/max_abs":8.8125,"train/train/tensor_act_model_layers_72_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/std":0.031005859375,"train/train/tensor_act_model_layers_13_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/std":0.0001265615649720735,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_54/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/std":0.039306640625,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/max_abs":0.00023174285888671875,"train/train/tensor_act_model_layers_1/mean":-0.020477294921875,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_v_proj/mean":0.001338958740234375,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_mlp_gate_proj_weight/max_abs":0.1337890625,"train/train/tensor_act_model_layers_52_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/norm":4.6875,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/std":0.031982421875,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/std":3.6332622211453005e-05,"train/train/tensor_act_model_layers_21_mlp_down_proj/max_abs":0.5234375,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/norm":0.01920630028516677,"train/train/tensor_act_model_layers_13_mlp_down_proj/std":0.04419147838102782,"train/train/layer__model_layers_8/param/norm":19.574134170267275,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/max_abs":0.1767578125,"train/train/tensor_act_model_layers_21_self_attn/std":0.08374116573040855,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/mean":-0.00010776519775390625,"train/train/tensor_act_model_layers_50_post_attention_layernorm/max_abs":6.09375,"train/train/tensor_grad_model_norm_weight/mean":-0.002910614013671875,"train/train/tensor_act_model_layers_90_mlp_up_proj/mean":-0.0015888214111328125,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/max_abs":0.251953125,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/max_abs":0.11767578125,"train/train/tensor_act_model_layers_55_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/norm":6.375,"train/train/tensor_act_model_layers_56_post_attention_layernorm/mean":-0.0015401840209960938,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/mean":5.817413330078125e-05,"train/train/tensor_act_model_layers_51_self_attn_o_proj/max_abs":1.2734375,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/max_abs":0.0002689361572265625,"train/train/tensor_act_model_layers_89_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_mlp/norm":206.55383678351737,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/mean":0.0002346038818359375,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_0/act/mean":-0.011063030787876673,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/max_abs":0.10107421875,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/mean":4.5736669562757015e-08,"train/train/tensor_act_model_layers_20_mlp_down_proj/std":0.03594987744548412,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/max_abs":0.00013446807861328125,"train/train/tensor_act_model_layers_77_self_attn_k_proj/norm":6124.114922803643,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/mean":-4.0675513446331024e-07,"train/train/tensor_act_model_layers_42_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_89_self_attn/norm":1023.0088243017645,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/std":1.5148929883340616e-05,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/max_abs":0.0004673004150390625,"train/train/tensor_act_model_layers_2_self_attn/norm":224.34503850420992,"train/train/tensor_act_model_layers_63_self_attn_k_proj/mean":0.00435638427734375,"train/train/tensor_param_model_layers_59_mlp_gate_proj_weight/mean":2.956390380859375e-05,"train/train/layer__model_layers_66/param/norm":23.628864597934026,"train/train/tensor_act_model_layers_25_mlp_up_proj/norm":2065.8685696609027,"train/train/tensor_act_model_layers_48_self_attn_k_proj/std":0.7460947554766286,"train/train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/norm":0.011495611571486255,"train/train/layer_model_layers_53/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/std":0.0283203125,"train/train/tensor_act_model_layers_9/std":1.289074942499857,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_57/grad/mean":-2.375272118636115e-07,"train/train/tensor_act_model_layers_69/mean":0.00763702392578125,"train/train/tensor_grad_model_layers_15_mlp_gate_proj_weight/std":4.102636742540252e-05,"train/train/tensor_act_model_layers_45_self_attn_q_proj/norm":4630.10575992962,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/mean":6.984919309616089e-09,"train/train/tensor_act_model_layers_27/norm":7224.737660666179,"train/train/layer_model_layers_1/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_k_proj/max_abs":5,"train/train/tensor_act_model_layers_50_mlp/mean":0.00017714500427246094,"train/train/tensor_act_model_layers_39_self_attn/norm":601.3627344159112,"train/train/tensor_act_model_layers_76/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/std":4.785669579587005e-05,"train/train/layer_model_layers_56/act/max_abs":9.875,"train/train/tensor_grad_model_layers_32_mlp_gate_proj_weight/std":4.377014775455311e-05,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/mean":-1.9073486328125e-05,"train/train/layer_model_layers_22/grad/mean":-2.3945596493350547e-08,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/mean":1.7257407307624817e-06,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_10_self_attn_k_proj/max_abs":5.28125,"train/train/tensor_act_model_layers_10_self_attn_v_proj/mean":-0.0012035369873046875,"train/train/tensor_act_model_layers_84_mlp_up_proj/std":0.5449254732713884,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/norm":0.013737487030283504,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_v_proj/mean":-0.00567626953125,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/std":3.23362433815675e-05,"train/train/tensor_grad_model_layers_84_mlp_gate_proj_weight/std":8.072221417951747e-05,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/max_abs":0.0003528594970703125,"train/train/tensor_act_model_layers_13_self_attn_o_proj/max_abs":0.6953125,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/mean":1.5762634575366974e-07,"train/train/tensor_act_model_layers_1_post_attention_layernorm/norm":5792.600708010049,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_mlp_gate_proj_weight/max_abs":0.000492095947265625,"train/train/tensor_act_model_layers_6_post_attention_layernorm/std":1.0000002174637976,"train/train/tensor_act_model_layers_39/std":1.2421881057929716,"train/train/layer_model_layers_7/grad/norm":0.03900479752445617,"train/train/tensor_act_model_layers_36_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_o_proj/norm":417.40750028311476,"train/train/tensor_act_model_layers_45_input_layernorm/mean":-0.0119476318359375,"train/train/layer_model_layers_53/act/mean":-0.0017559868948800223,"train/train/tensor_act_model_layers_16_post_attention_layernorm/std":1.0000012014053996,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_gate_proj/norm":2025.115149712896,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/std":0.061279296875,"train/train/tensor_grad_model_layers_92_mlp_gate_proj_weight/std":9.574601090166481e-05,"train/train/tensor_act_model_layers_70_mlp/norm":765.3281141027179,"train/train/tensor_act_model_layers_85_mlp/mean":0.00223541259765625,"train/train/tensor_act_/mean":2.000778168439865,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/norm":6.59375,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/mean":-2.777576446533203e-05,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/mean":-5.278736352920532e-06,"train/train/layer_model_layers_44/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_mlp_gate_proj_weight/norm":6.78125,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/mean":0.000171661376953125,"train/train/tensor_act_model_layers_78_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_gate_proj_weight/std":0.040283203125,"train/train/tensor_act_model_layers_26_self_attn_v_proj/std":0.33398441989954847,"train/train/tensor_act_model_layers_69_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_gate_proj/std":0.38281252067916194,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/max_abs":0.1767578125,"train/train/tensor_act_model_embed_tokens/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/max_abs":0.146484375,"train/train/tensor_act_model_layers_80_self_attn_q_proj/mean":0.0731201171875,"train/train/tensor_act_model_layers_40_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_mlp_gate_proj_weight/std":4.3940394894264006e-05,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/norm":0.001189145832268406,"train/train/tensor_act_model_layers_57_self_attn_k_proj/max_abs":5.9375,"train/train/layer__model_layers_25/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/mean":3.686174750328064e-06,"train/train/tensor_act_model_layers_55/norm":7481.654180662038,"train/train/tensor_param_model_layers_39_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn/norm":587.5135877209094,"train/train/tensor_act_model_layers_90_self_attn_q_proj/std":1.1679759136988155,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/norm":5.9375,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_93/act/norm":26915.247419024177,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/norm":0.013214710389405056,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_0/grad/norm":0.3266359077528467,"train/train/tensor_grad_model_layers_79_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/norm":5.0625,"train/train/tensor_grad_model_layers_22_mlp_gate_proj_weight/mean":1.0826624929904938e-08,"train/train/layer_model_layers_31/act/norm":13498.221212206841,"train/train/tensor_grad_model_layers_8_mlp_gate_proj_weight/norm":0.01578006688017083,"train/train/tensor_act_model_layers_80_self_attn_v_proj/norm":2890.4613914189326,"train/train/tensor_param_model_layers_83_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/norm":6.28125,"train/train/tensor_act_model_layers_27_self_attn_v_proj/mean":0.003932952880859375,"train/train/tensor_grad_model_layers_24_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/norm":7,"train/train/tensor_act_model_layers_31_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_2_mlp/max_abs":1.5546875,"train/train/layer_model_layers_4/act/max_abs":8.8125,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/std":0.04736328125,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/max_abs":0.000293731689453125,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_mlp/max_abs":0.46484375,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn/norm":1117.325149057583,"train/train/tensor_act_model_layers_9_mlp_up_proj/mean":0.001583099365234375,"train/train/tensor_act_model_layers_62_post_attention_layernorm/std":1.0000005488412484,"train/train/tensor_act_model_layers_7_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/norm":4.3125,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn_o_proj/mean":-0.00021839141845703125,"train/train/tensor_act_model_layers_77_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/std":0.03173828125,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/std":6.085448345504242e-05,"train/train/tensor_param_model_layers_16_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/mean":-0.000194549560546875,"train/train/layer_model_layers_69/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/max_abs":0.287109375,"train/train/tensor_act_model_layers_16_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/norm":7.1875,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/std":8.556986482196563e-05,"train/train/layer_model_layers_49/act/max_abs":9.8125,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/max_abs":0.2353515625,"train/train/tensor_param_model_layers_50_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/mean":-7.486343383789062e-05,"train/train/tensor_grad_model_layers_30_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48/max_abs":9.75,"train/train/tensor_act_model_layers_5_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/max_abs":0.171875,"train/train/tensor_act_model_layers_40_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn/norm":267.57110567920625,"train/train/tensor_act_model_layers_1_mlp_gate_proj/norm":2915.046629454494,"train/train/tensor_act_model_layers_44_input_layernorm/mean":-0.012939453125,"train/train/tensor_act_model_layers_60_mlp/max_abs":0.7578125,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/norm":3.109375,"train/train/tensor_act_model_layers_93_mlp/mean":-0.019775390625,"train/train/tensor_act_model_layers_32_self_attn_v_proj/norm":2105.7054677577466,"train/train/tensor_act_model_layers_92_mlp/norm":3724.876187052388,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/mean":0.00017070770263671875,"train/train/tensor_act_model_layers_30_mlp_gate_proj/norm":2231.6547572493255,"train/train/tensor_act_model_layers_52_self_attn_k_proj/max_abs":5.75,"train/train/tensor_act_model_layers_46_mlp_up_proj/norm":2707.017082235649,"train/train/tensor_act_model_layers_46_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31/norm":7210.753058481518,"train/train/tensor_act_model_layers_82_self_attn/mean":0.0020732879638671875,"train/train/tensor_act_model_layers_80_mlp/norm":1111.8647286990276,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/norm":6.875,"train/train/tensor_act_model_layers_27/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/norm":6.03125,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/std":0.05810546875,"train/train/tensor_act_model_layers_32_self_attn_q_proj/mean":-0.02862548828125,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/std":3.224829817274061e-05,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/std":0.046142578125,"train/train/tensor_act_model_layers_32_self_attn/mean":0.0005407333374023438,"train/train/tensor_act_model_layers_54_self_attn_q_proj/max_abs":5,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/max_abs":0.00052642822265625,"train/train/tensor_act_model_layers_62_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/mean":1.1129304766654968e-07,"train/train/layer_model_layers_38/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/max_abs":0.1513671875,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/std":6.320160963632316e-05,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn/max_abs":0.8046875,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/std":0.045166015625,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/max_abs":0.25390625,"train/train/tensor_act_model_layers_42_self_attn_q_proj/std":0.9179694723572831,"train/train/tensor_act_model_layers_87_input_layernorm/max_abs":5.6875,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/max_abs":0.16015625,"train/train/tensor_act_model_layers_13_post_attention_layernorm/max_abs":5.875,"train/train/tensor_act_model_layers_46_input_layernorm/std":1.0000003566964824,"train/train/tensor_act_model_layers_74_mlp/mean":0.0016307830810546875,"train/train/tensor_act_model_layers_40_self_attn_k_proj/std":0.7783222766837078,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/max_abs":0.1318359375,"train/train/tensor_act_model_layers_87/norm":12083.96323187237,"train/train/layer_model_layers_35/act/norm":13438.643340259621,"train/train/tensor_act_model_layers_51_mlp/norm":426.6580840913031,"train/train/tensor_act_model_layers_49_mlp/mean":0.001071929931640625,"train/train/tensor_act_model_layers_59_mlp_gate_proj/norm":3274.5093403698897,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/std":0.04736328125,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_19_mlp_gate_proj/mean":0.002346038818359375,"train/train/tensor_act_model_layers_62_self_attn/std":0.09497983871107281,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/max_abs":0.154296875,"train/train/tensor_act_model_layers_45_self_attn_v_proj/mean":-0.002399444580078125,"train/train/tensor_act_model_layers_58_input_layernorm/max_abs":5.9375,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_22/param/mean":0.001440110705014138,"train/train/tensor_grad_model_layers_91_mlp_gate_proj_weight/std":9.491345099692773e-05,"train/train/tensor_act_model_layers_12_self_attn/norm":442.59687073856117,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/std":2.9603630743963807e-05,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/mean":-4.263129085302353e-07,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/mean":-2.1725893020629883e-05,"train/train/tensor_act_model_layers_69_self_attn_q_proj/std":0.9160222642216703,"train/train/tensor_grad_model_layers_13_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_56/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn/max_abs":0.283203125,"train/train/tensor_act_model_layers_59_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/max_abs":0.0002727508544921875,"train/train/tensor_act_model_layers_20_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_mlp_down_proj/max_abs":0.5,"train/train/tensor_act_model_layers_60_mlp_up_proj/max_abs":2.734375,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/max_abs":0.1474609375,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/mean":0.0001201629638671875,"train/train/tensor_act_model_layers_57_mlp_gate_proj/mean":0.012603759765625,"train/train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/max_abs":0.0002593994140625,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/mean":-3.1781382858753204e-07,"train/train/tensor_act_model_layers_47_mlp_down_proj/mean":0.0002276897430419922,"train/train/tensor_grad_model_layers_55_mlp_gate_proj_weight/std":6.0517596519339916e-05,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_18/act/max_abs":8.4375,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/norm":0.017285992445897635,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_64/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/norm":0.01409188617511458,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_k_proj/norm":5286.370009962332,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/norm":0.000662547686434801,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_0/grad/max_abs":0.008056640625,"train/train/tensor_act_model_layers_8_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/std":0.05615234375,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/max_abs":0.1279296875,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/norm":0.030443330088918447,"train/train/tensor_act_model_layers_26_mlp_up_proj/max_abs":2.203125,"train/train/layer_model_layers_15/act/max_abs":8.4375,"train/train/tensor_grad_model_layers_45_mlp_gate_proj_weight/max_abs":0.0004367828369140625,"train/train/tensor_act_model_layers_89_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_mlp_gate_proj_weight/mean":1.2404052540659904e-07,"train/train/tensor_act_model_layers_43_self_attn_q_proj/norm":5417.856063757368,"train/train/tensor_act_model_layers_50/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_mlp_gate_proj_weight/norm":7,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_mlp_gate_proj_weight/norm":0.015093989204501569,"train/train/tensor_act_model_layers_90_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_70/param/std":0.057480150788055184,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/mean":-3.461173037067056e-08,"train/train/tensor_act_model_layers_10_self_attn_q_proj/norm":5160.263893247006,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/std":7.62019639886597e-05,"train/train/tensor_act_model_layers_11_mlp_up_proj/std":0.23046892846865805,"train/train/tensor_act_model_layers_87_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp/mean":0.0006389617919921875,"train/train/tensor_act_model_layers_83_self_attn_q_proj/mean":0.0650634765625,"train/train/tensor_act_model_layers_67_self_attn_k_proj/max_abs":6.03125,"train/train/tensor_act_model_layers_17_post_attention_layernorm/mean":-0.027740478515625,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/norm":3.484375,"train/train/tensor_act_model_layers_91_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn_o_proj/max_abs":1.4140625,"train/train/tensor_act_model_layers_79_mlp_down_proj/std":0.18139703397983267,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/mean":-0.00019359588623046875,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/mean":2.8133392333984375e-05,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/max_abs":0.19921875,"train/train/tensor_act_model_layers_50_self_attn_o_proj/max_abs":1.640625,"train/train/tensor_act_model_layers_27/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/std":4.587106284214833e-05,"train/train/layer_model_layers_10/act/norm":13789.051616019693,"train/train/tensor_act_model_layers_80_self_attn/std":0.2243809738919304,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/std":0.033935546875,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/mean":0.00014781951904296875,"train/train/tensor_act_model_layers_1/max_abs":8.3125,"train/train/layer_model_layers_63/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/mean":-0.00011682510375976562,"train/train/tensor_param_model_layers_32_mlp_gate_proj_weight/norm":5.125,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/mean":-5.820766091346741e-06,"train/train/tensor_act_model_layers_37_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/std":0.03125,"train/train/tensor_act_model_layers_58_self_attn_v_proj/std":0.38916111271945836,"train/train/tensor_act_model_layers_41_post_attention_layernorm/std":1.0000002955784584,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/max_abs":0.2333984375,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/mean":2.8371810913085938e-05,"train/train/tensor_param_model_layers_58_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/mean":0.000637054443359375,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/max_abs":0.12060546875,"train/train/tensor_act_model_layers_87_input_layernorm/std":1.0000007442137717,"train/train/tensor_act_model_layers_82_self_attn/std":0.21460028722419217,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/mean":3.116438165307045e-07,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_v_proj/std":0.28222831895368305,"train/train/tensor_act_model_layers_63_self_attn_v_proj/max_abs":2.3125,"train/train/tensor_act_model_layers_27_mlp_up_proj/max_abs":1.8203125,"train/train/tensor_act_model_layers_45/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/mean":-7.915496826171875e-05,"train/train/tensor_act_model_layers_61_self_attn_v_proj/max_abs":4.84375,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_87_mlp_down_proj/std":0.3125001600710265,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/max_abs":0.12451171875,"train/train/tensor_act_model_layers_87_self_attn/max_abs":2.375,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/norm":0.010957765537619877,"train/train/tensor_grad_model_layers_25_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/norm":0.018228065366629465,"train/train/tensor_grad_model_layers_29_mlp_gate_proj_weight/std":4.270924746699767e-05,"train/train/tensor_act_model_layers_8_input_layernorm/std":1.0000004260799873,"train/train/tensor_act_model_layers_61_mlp_down_proj/mean":-6.628036499023438e-05,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/max_abs":0.26953125,"train/train/layer_model_layers_81/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_v_proj/norm":2352.1835339624804,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/norm":6.5,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/norm":5.25,"train/train/tensor_act_model_layers_84_self_attn/std":0.22070438331005232,"train/train/tensor_act_model_layers_9_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_v_proj/mean":-0.00347137451171875,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_q_proj/norm":5783.454873901665,"train/train/tensor_act_model_layers_41_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/mean":0.0001544952392578125,"train/train/tensor_act_model_layers_40_self_attn_k_proj/mean":0.019927978515625,"train/train/tensor_act_model_layers_44_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_72/param/norm":24.00699767776054,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/max_abs":0.00115966796875,"train/train/tensor_act_model_layers_92_self_attn/mean":-0.001979827880859375,"train/train/tensor_act_model_layers_68_mlp_down_proj/mean":-0.000885009765625,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/norm":0.01554384614203783,"train/train/tensor_act_model_layers_14_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_56/param/max_abs":1,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/std":6.752589831862462e-05,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/norm":2.984375,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/max_abs":0.0001430511474609375,"train/train/tensor_act_model_layers_20_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/mean":-7.05718994140625e-05,"train/train/layer_model_layers_6/act/std":0.6604534264179892,"train/train/tensor_param_model_layers_55_mlp_gate_proj_weight/norm":6.125,"train/train/tensor_act_model_layers_18_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_27/param/mean":0.0015472614449011555,"train/train/tensor_param_model_layers_92_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_88/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/mean":6.914138793945312e-05,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/max_abs":0.2197265625,"train/train/tensor_act_model_layers_6_post_attention_layernorm/max_abs":5.09375,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_64/act/mean":-0.0092391094991139,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp_down_proj/mean":-0.0006780624389648438,"train/train/tensor_act_model_layers_2_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/norm":7.03125,"train/train/tensor_act_model_layers_92_mlp/mean":0.0088958740234375,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer__model_layers_43/param/mean":0.0014098073688572543,"train/train/tensor_act_model_layers_87_self_attn_o_proj/norm":1404.9770829646761,"train/train/layer_model_layers_35/grad/std":4.208242455206433e-05,"train/train/tensor_act_model_layers_65_input_layernorm/max_abs":5.6875,"train/train/tensor_grad_model_layers_31_mlp_gate_proj_weight/max_abs":0.000553131103515625,"train/train/tensor_act_model_layers_12_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_v_proj/std":0.3535157476824326,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/mean":-3.680586814880371e-06,"train/train/tensor_act_model_layers_67_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/norm":0.004988994033659088,"train/train/tensor_act_model_layers_61_post_attention_layernorm/mean":0.00022554397583007812,"train/train/tensor_act_model_layers_58_self_attn_o_proj/mean":0.0001899339258670807,"train/train/tensor_act_model_layers_89_post_attention_layernorm/max_abs":5.5625,"train/train/layer_model_layers_61/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_self_attn/std":0.09576541487301234,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/std":0.03466796875,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/mean":-8.177757263183594e-05,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/max_abs":0.271484375,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/mean":2.8476642910391092e-06,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn/std":0.22632108612640064,"train/train/tensor_act_model_layers_34_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn/max_abs":0.55859375,"train/train/layer_model_layers_52/act/mean":-0.001425487654549735,"train/train/tensor_act_model_layers_17_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp/max_abs":11.75,"train/train/tensor_act_model_layers_0_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_down_proj/max_abs":0.44140625,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/norm":0.0012996982047918976,"train/train/tensor_act_model_layers_59_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_input_layernorm/std":1.0000015459942788,"train/train/tensor_act_model_layers_16_input_layernorm/norm":5792.604858399038,"train/train/tensor_param_model_layers_54_mlp_gate_proj_weight/std":0.033935546875,"train/train/tensor_act_model_layers_36_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/norm":3.515625,"train/train/tensor_act_model_layers_79_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_71/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp_down_proj/norm":532.8931969174981,"train/train/tensor_grad_model_layers_80_mlp_gate_proj_weight/mean":3.725290298461914e-09,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/mean":9.560026228427887e-07,"train/train/tensor_act_model_layers_70_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/max_abs":0.0001659393310546875,"train/train/tensor_act_model_layers_55_self_attn_q_proj/std":0.8359421079276885,"train/train/layer_model_layers_30/act/mean":0.0030813046864100863,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/max_abs":0.0003452301025390625,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn_q_proj/mean":-0.0234375,"train/train/tensor_act_model_layers_17_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_75/grad/norm":0.06323429225950122,"train/train/tensor_act_model_layers_89_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_post_attention_layernorm/mean":-0.0141143798828125,"train/train/tensor_act_model_layers_39_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_down_proj/norm":222.48812308687425,"train/train/tensor_act_model_layers_46_mlp_gate_proj/norm":2704.069598005824,"train/train/tensor_act_lm_head/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/mean":6.062909960746765e-06,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/mean":3.064633347094059e-08,"train/train/tensor_act_model_layers_82_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_v_proj/std":0.3100597665689,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/max_abs":0.251953125,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/max_abs":0.000423431396484375,"train/train/tensor_act_model_layers_11_self_attn_v_proj/norm":1858.5346572045837,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/norm":0.01923004593772777,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/max_abs":0.11669921875,"train/train/layer__model_layers_77/param/max_abs":1,"train/train/tensor_param_model_layers_44_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/max_abs":0.2392578125,"train/train/tensor_act_model_layers_3_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_input_layernorm/max_abs":5.90625,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/norm":0.00820152460108769,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/max_abs":0.248046875,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_64_self_attn_k_proj/mean":-0.049560546875,"train/train/tensor_act_model_layers_61_self_attn_k_proj/norm":4618.153899440812,"train/train/tensor_act_model_layers_27_self_attn_q_proj/norm":5526.934123558238,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_34/max_abs":9.3125,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/std":0.00011555509916564499,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/norm":3.90625,"train/train/tensor_act_model_layers_67_input_layernorm/max_abs":5.5,"train/train/tensor_act_model_layers_14_mlp/std":0.03558366012116262,"train/train/tensor_act_model_layers_8_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_k_proj/mean":0.00507354736328125,"train/train/tensor_act_model_layers_74_self_attn_q_proj/max_abs":6.65625,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/norm":4.5,"train/train/tensor_param_model_layers_10_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_54/act/max_abs":9.9375,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_input_layernorm/max_abs":5.6875,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/max_abs":0.000396728515625,"train/train/tensor_param_model_layers_86_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/norm":0.015754993426746196,"train/train/layer_model_layers_22/act/std":0.6296585425295185,"train/train/tensor_act_model_layers_6_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/norm":6.5625,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/norm":0.002331338743682483,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/max_abs":0.0004634857177734375,"train/train/layer_model_layers_52/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/max_abs":0.0002536773681640625,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/max_abs":0.203125,"train/train/tensor_act_model_layers_30_mlp/mean":0.0005731582641601562,"train/train/tensor_act_model_layers_15_mlp_gate_proj/std":0.2409671694490341,"train/train/tensor_act_model_layers_13_self_attn_q_proj/max_abs":6.8125,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/mean":-2.130400389432907e-07,"train/train/layer_model_layers_83/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66/mean":0.00594329833984375,"train/train/tensor_act_model_layers_24/max_abs":8.5625,"train/train/tensor_act_model_layers_38_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn_v_proj/max_abs":2.1875,"train/train/tensor_act_model_layers_22_post_attention_layernorm/std":1.0000014081587414,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/std":7.106423514748529e-05,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/std":4.4346932059302946e-05,"train/train/layer__model_layers_0/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_q_proj/std":0.9404312234788753,"train/train/tensor_act_model_layers_32_input_layernorm/mean":-0.0129241943359375,"train/train/tensor_grad_model_layers_18_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_92/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp_down_proj/max_abs":0.447265625,"train/train/tensor_act_model_layers_6_self_attn_q_proj/max_abs":6.03125,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_11/act/std":0.6317137330994937,"train/train/layer__model_layers_53/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn/std":0.07946882225144182,"train/train/tensor_act_model_layers_72_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_input_layernorm/mean":-0.003253936767578125,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/mean":-2.6887282729148865e-06,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_mlp_down_proj/std":0.06384313031042402,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_4_mlp_down_proj/norm":387.20331548691627,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/mean":6.763730198144913e-08,"train/train/tensor_grad_model_layers_64_mlp_gate_proj_weight/std":7.105630157698714e-05,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_mlp_gate_proj_weight/mean":2.8172507882118225e-08,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/std":2.0672117796938343e-05,"train/train/tensor_act_model_layers_65_self_attn_q_proj/norm":5899.711327628441,"train/train/layer__model_layers_79/param/norm":24.097944673768342,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/std":0.038330078125,"train/train/tensor_act_model_layers_50/std":1.2656255610929823,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/norm":7.25,"train/train/tensor_act_model_layers_37_mlp_down_proj/norm":308.2524722240277,"train/train/tensor_act_model_layers_89_mlp_down_proj/max_abs":2.421875,"train/train/tensor_act_model_layers_26_input_layernorm/norm":5792.6112060596715,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/mean":-1.395528670400381e-07,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/std":3.138163675183973e-05,"train/train/tensor_param_model_layers_12_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_81_self_attn_v_proj/max_abs":3.515625,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/std":0.0439453125,"train/train/layer__model_layers_26/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_40/param/frac_near_user_limit":0,"train/train/layer__model_layers_48/param/norm":21.745868860544523,"train/train/tensor_act_model_layers_28_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/norm":0.00450519461620499,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/std":7.390694249866163e-05,"train/train/tensor_act_model_layers_25/mean":-0.021759033203125,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/std":0.053955078125,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/std":1.0546875423855242,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/norm":6.40625,"train/train/tensor_act_model_layers_51_self_attn_q_proj/max_abs":6.375,"train/train/tensor_act_model_layers_39_self_attn/std":0.10388278092247015,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/max_abs":0.0004787445068359375,"train/train/tensor_act_model_layers_65_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67/max_abs":10.875,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/std":0.0264892578125,"train/train/tensor_act_model_layers_87_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_input_layernorm/norm":5792.608642579818,"train/train/tensor_act_model_layers_68_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_87/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_o_proj/mean":-0.00030541419982910156,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_mlp_down_proj/mean":-0.00019252300262451172,"train/train/tensor_grad_model_layers_21_mlp_gate_proj_weight/max_abs":0.00090789794921875,"train/train/tensor_act_model_layers_72_input_layernorm/mean":0.005481719970703125,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_gate_proj_weight/mean":-0.000194549560546875,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/mean":-6.79574441164732e-09,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/max_abs":0.10791015625,"train/train/tensor_act_model_layers_33_self_attn/max_abs":0.953125,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/norm":0.04103560188265731,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/max_abs":0.126953125,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/norm":0.017006480026792364,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/mean":-5.667097866535187e-07,"train/train/tensor_act_model_layers_10_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/std":8.411560074408001e-05,"train/train/tensor_act_model_layers_82_mlp_gate_proj/mean":-0.0113372802734375,"train/train/tensor_act_model_layers_75_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/mean":0.0001735687255859375,"train/train/tensor_act_model_layers_71_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_mlp_gate_proj_weight/std":7.880617234238615e-05,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/mean":8.630752563476562e-05,"train/train/tensor_act_model_layers_50_mlp/norm":408.53799970398313,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/norm":0.017742763501538424,"train/train/tensor_act_model_layers_50_self_attn/norm":541.1735998029751,"train/train/tensor_act_model_layers_48_input_layernorm/mean":-0.011138916015625,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/std":0.0361328125,"train/train/tensor_param_model_layers_77_mlp_gate_proj_weight/norm":7.5,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_mlp/std":0.04419147838102782,"train/train/tensor_act_model_layers_28_mlp_gate_proj/std":0.26171880549013915,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/norm":2.640625,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/std":1.7207417230555437e-05,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_21/act/max_abs":8.5,"train/train/tensor_act_model_layers_34_mlp_gate_proj/mean":-0.0045166015625,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/mean":1.1138617992401123e-06,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/std":0.032958984375,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_mlp_gate_proj_weight/mean":-2.3096799850463867e-07,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_1_mlp_down_proj/norm":1323.0838183677952,"train/train/tensor_act_model_layers_75_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/std":8.936961407687221e-05,"train/train/layer__model_layers_42/param/std":0.05328521975048035,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_q_proj/mean":0.018157958984375,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/max_abs":0.00014019012451171875,"train/train/tensor_act_model_layers_52_self_attn_o_proj/norm":241.709215287855,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/mean":-4.682806320488453e-08,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/norm":0.024599363924735055,"train/train/tensor_grad_model_layers_67_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_o_proj/max_abs":1.796875,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/max_abs":0.0003566741943359375,"train/train/tensor_act_model_layers_19_self_attn/max_abs":1.15625,"train/train/tensor_act_model_layers_41_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn_q_proj/std":0.892580559657301,"train/train/tensor_act_model_layers_34_mlp_gate_proj/std":0.28710962963741715,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/max_abs":0.0003376007080078125,"train/train/tensor_act_model_layers_90_input_layernorm/mean":0.007720947265625,"train/train/layer_model_layers_84/act/mean":0.001501491027218955,"train/train/tensor_act_model_layers_89_input_layernorm/std":1.0000008797210962,"train/train/tensor_act_model_layers_0_mlp_down_proj/mean":-0.0223388671875,"train/train/tensor_act_model_layers_46_post_attention_layernorm/std":1.0000003068707413,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn/norm":557.2084066712753,"train/train/tensor_act_model_layers_71_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/std":0.02783203125,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/std":0.00010039560801438636,"train/train/layer_model_layers_32/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/norm":0.0015160482349662898,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/max_abs":0.23046875,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/mean":0.000335693359375,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/std":0.0380859375,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/max_abs":0.000247955322265625,"train/train/tensor_act_model_layers_62/std":1.332037841595653,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_q_proj/std":0.6845724347961876,"train/train/layer_model_layers_47/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/std":9.517480714646266e-05,"train/train/tensor_act_model_layers_52_self_attn_v_proj/max_abs":1.859375,"train/train/tensor_param_model_layers_84_mlp_gate_proj_weight/norm":8.0625,"train/train/tensor_grad_model_layers_10_mlp_gate_proj_weight/std":4.1491731107790105e-05,"train/train/tensor_act_model_layers_60/mean":-0.0014495849609375,"train/train/tensor_act_model_layers_34_mlp_down_proj/norm":272.3008345377486,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/std":8.135094863380703e-05,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/max_abs":0.00014972686767578125,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/norm":0.014384174839792652,"train/train/tensor_act_model_layers_82_self_attn_o_proj/mean":0.0020732879638671875,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_v_proj/norm":2976.829067879069,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_26/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/std":0.040771484375,"train/train/tensor_act_model_layers_88_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_mlp_gate_proj_weight/norm":0.019512835404802063,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/max_abs":0.236328125,"train/train/tensor_act_model_layers_43_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_mlp_down_proj/norm":1217.726066880047,"train/train/tensor_param_model_layers_29_mlp_gate_proj_weight/norm":5.03125,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/mean":-3.147125244140625e-05,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_70/param/mean":0.001617407835961876,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_3/param/std":0.04750754324290828,"train/train/layer_model_layers_46/grad/max_abs":0.00147247314453125,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/max_abs":0.234375,"train/train/layer__model_layers_66/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/std":0.03662109375,"train/train/tensor_act_model_layers_77_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/mean":-0.00010967254638671875,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/mean":0.00016689300537109375,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/norm":0.0009340128703518912,"train/train/tensor_act_model_layers_89_self_attn_o_proj/norm":1023.0088243017645,"train/train/tensor_act_model_layers_63_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_self_attn/norm":459.51288475145,"train/train/tensor_act_model_layers_43_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp/std":0.045532413565696546,"train/train/tensor_act_model_layers_87_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_58/act/mean":0.006953612502132144,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_post_attention_layernorm/mean":0.005657196044921875,"train/train/tensor_act_model_layers_56_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/std":0.028564453125,"train/train/layer_model_layers_83/act/std":0.8096122268814477,"train/train/tensor_act_model_layers_7_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/mean":8.41100700199604e-08,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/norm":0.004805210480488644,"train/train/tensor_act_model_layers_45_mlp_gate_proj/std":0.322755045318902,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_mlp_gate_proj_weight/mean":-7.915496826171875e-05,"train/train/tensor_act_model_layers_45_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73/max_abs":10.9375,"train/train/tensor_act_model_layers_63_mlp_down_proj/max_abs":0.80859375,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/norm":5.9375,"train/train/layer_model_layers_93/grad/norm":0.07692304498105668,"train/train/tensor_param_model_layers_76_mlp_gate_proj_weight/max_abs":0.19140625,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_mlp_down_proj/max_abs":0.5703125,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_93/grad/std":9.492041566595106e-05,"train/train/tensor_act_model_layers_32_mlp_down_proj/std":0.04486098303452571,"train/train/tensor_param_model_layers_66_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/max_abs":9.393692016601562e-05,"train/train/tensor_act_model_layers_57_self_attn_v_proj/max_abs":2.953125,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_37_mlp_gate_proj/mean":0.001873016357421875,"train/train/tensor_act_model_layers_22_self_attn_o_proj/norm":297.2261457100618,"train/train/tensor_param_model_layers_11_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_73/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/norm":3.234375,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/norm":0.03543236702372638,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/max_abs":0.00010919570922851562,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/std":8.580145346323399e-06,"train/train/layer_model_layers_49/grad/std":5.360290664674967e-05,"train/train/tensor_act_model_layers_85_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/mean":0.000308990478515625,"train/train/tensor_act_model_layers_39_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/mean":-8.916854858398438e-05,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/norm":0.02316595767729288,"train/train/tensor_act_model_layers_84_self_attn_k_proj/mean":-0.0026302337646484375,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/std":7.527702651017481e-05,"train/train/tensor_act_model_layers_56_self_attn/std":0.11096270598815598,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/std":0.042724609375,"train/train/tensor_act_model_layers_0/mean":-0.023468017578125,"train/train/tensor_param_model_layers_86_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_down_proj/max_abs":1.2265625,"train/train/tensor_act_model_layers_34_mlp/max_abs":0.3515625,"train/train/tensor_act_model_layers_61_self_attn_k_proj/max_abs":6.71875,"train/train/tensor_act_model_layers_45_post_attention_layernorm/norm":5792.602539063035,"train/train/tensor_act_model_layers_83_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_52/act/max_abs":9.9375,"train/train/tensor_act_model_layers_47/mean":-0.014251708984375,"train/train/tensor_param_model_layers_39_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_o_proj/mean":0.00295257568359375,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/mean":-5.98374754190445e-08,"train/train/layer__model_layers_35/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/norm":6.125,"train/train/tensor_act_model_layers_5_self_attn_v_proj/std":0.30664065491502873,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/max_abs":0.1962890625,"train/train/tensor_act_model_layers_19_self_attn_q_proj/mean":-0.0092010498046875,"train/train/tensor_act_model_layers_92_mlp_down_proj/std":0.6425811083051888,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/norm":0.003064871270199377,"train/train/layer__model_layers_34/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/mean":-1.843273639678955e-05,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn_q_proj/std":1.2031254474217958,"train/train/tensor_grad_model_layers_13_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_post_attention_layernorm/max_abs":5.40625,"train/train/tensor_act_model_layers_76/mean":0.005879878997802734,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/mean":7.309019565582275e-06,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/mean":-2.0579318515956402e-07,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/max_abs":0.00064849853515625,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/std":4.2164812761566746e-05,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/std":5.140119547456863e-05,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_k_proj/max_abs":5.65625,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/norm":0.0006564438968411645,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/norm":0.0011694084973121134,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/frac_near_dtype_limit":0,"_timestamp":1.786255032070748e+09,"train/train/tensor_act_model_layers_7_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_v_proj/norm":1754.9619206791647,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_v_proj/max_abs":3.265625,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/mean":1.0669231414794922e-05,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/max_abs":0.00012302398681640625,"train/train/tensor_act_model_layers_76_input_layernorm/norm":5792.607177736871,"train/train/layer_model_layers_39/grad/norm":0.04275443649517294,"train/train/tensor_param_model_layers_8_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_50_self_attn_q_proj/max_abs":5.1875,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/mean":7.124617695808411e-07,"train/train/tensor_act_model_layers_30/norm":7202.836873515226,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/norm":0.0006017858741535495,"train/train/tensor_act_model_layers_48_self_attn_q_proj/max_abs":5.875,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/norm":6.15625,"train/train/tensor_act_model_layers_24_mlp_up_proj/frac_near_user_limit":0,"train/train/layer__model_layers_51/param/max_abs":1,"train/train/layer__model_layers_40/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_down_proj/max_abs":0.404296875,"train/train/tensor_act_model_layers_26_mlp_gate_proj/norm":2076.8201816214564,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/std":6.4665570599906e-05,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/max_abs":0.2197265625,"train/train/layer__model_layers_0/param/mean":0.0015995781432819813,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_mlp_gate_proj_weight/max_abs":0.181640625,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/norm":0.0018459232958410534,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/norm":0.04157045302629869,"train/train/tensor_act_model_layers_32_mlp/max_abs":0.3671875,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/std":3.043064462901829e-05,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/std":0.0279541015625,"train/train/layer_model_layers_63/grad/max_abs":0.00118255615234375,"train/train/layer_model_layers_42/act/max_abs":9.3125,"train/train/tensor_act_model_layers_10_self_attn/max_abs":1.171875,"train/train/layer__model_layers_92/param/mean":0.00167261084974649,"train/train/tensor_act_model_layers_49_input_layernorm/std":1.0000003114109373,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/std":0.04150390625,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/max_abs":0.0012664794921875,"train/train/tensor_act_model_layers_68_post_attention_layernorm/std":1.0000019683483243,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/std":0.041015625,"train/train/layer__model_layers_63/param/max_abs":1,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/std":7.698661264920904e-06,"train/train/tensor_act_model_layers_55_input_layernorm/max_abs":5.84375,"train/train/layer_model_layers_37/act/norm":14144.676575231395,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/mean":2.1615996956825256e-06,"train/train/tensor_param_model_layers_15_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_3/grad/max_abs":0.0017547607421875,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_mlp_gate_proj_weight/mean":-1.7974525690078735e-07,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_17_input_layernorm/norm":5792.612670900581,"train/train/tensor_act_model_layers_70_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn/mean":0.0009241104125976562,"train/train/layer__model_layers_38/param/mean":0.0014990272462461,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/norm":0.016110766821526838,"train/train/tensor_act_model_layers_4_self_attn_o_proj/norm":798.2307543961442,"train/train/tensor_param_model_layers_31_mlp_gate_proj_weight/std":0.028076171875,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn/mean":-0.0003724396228790283,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/mean":-9.66247171163559e-08,"train/train/tensor_act_model_layers_48_self_attn_o_proj/std":0.08068960477653225,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/std":3.587109188038334e-05,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/std":0.04345703125,"train/train/tensor_act_model_layers_83/max_abs":11.25,"train/train/tensor_act_model_layers_57/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8/std":1.2890750133022648,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_44_self_attn_o_proj/mean":0.001178741455078125,"train/train/tensor_act_model_layers_68/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_v_proj/mean":0.0053863525390625,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_32/param/mean":0.001676502912167268,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/std":7.892288429390613e-05,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/max_abs":0.2080078125,"train/train/tensor_param_model_layers_47_mlp_gate_proj_weight/mean":-8.678436279296875e-05,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/norm":6.65625,"train/train/layer__model_layers_89/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/std":2.9607391438948295e-05,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/mean":-2.959277480840683e-07,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/mean":2.3283064365386963e-09,"train/train/tensor_param_model_layers_36_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/max_abs":0.197265625,"train/train/tensor_act_model_layers_52_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_62/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn_o_proj/max_abs":2.71875,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/std":1.924883044429793e-05,"train/train/tensor_act_model_layers_49_self_attn_v_proj/std":0.38671902537035047,"train/train/tensor_act_model_layers_56_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/std":0.030517578125,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/mean":-1.430744305253029e-07,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/max_abs":0.000518798828125,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp/mean":0.0002727508544921875,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/act/std":0.6227357635708751,"train/train/tensor_act_model_layers_23_post_attention_layernorm/mean":-0.0191650390625,"train/train/tensor_param_model_layers_86_mlp_gate_proj_weight/mean":0.00012111663818359375,"train/train/tensor_act_model_layers_30_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_mlp_gate_proj_weight/std":0.03173828125,"train/train/tensor_act_model_layers_81_post_attention_layernorm/std":1.000001637118227,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/max_abs":0.00043487548828125,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn/max_abs":1.9765625,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/max_abs":0.00012111663818359375,"train/train/tensor_act_model_layers_21_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/max_abs":0.12890625,"train/train/tensor_act_model_layers_74_input_layernorm/mean":0.00724029541015625,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/max_abs":0.000560760498046875,"train/train/tensor_act_model_layers_10_mlp/std":0.039551189106821226,"train/train/tensor_act_model_layers_62_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/norm":3.421875,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/max_abs":0.00023746490478515625,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/norm":5.09375,"train/train/tensor_act_model_layers_80_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/norm":0.023990621095164852,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/mean":-3.3574178814888e-07,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/norm":3.96875,"train/train/layer_model_layers_33/act/max_abs":9.1875,"train/train/tensor_param_model_layers_93_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/norm":0.023847568896084446,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/mean":-0.000530242919921875,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/max_abs":0.146484375,"train/train/tensor_act_model_layers_20/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_post_attention_layernorm/mean":-0.011810302734375,"train/train/tensor_act_model_layers_50_self_attn_k_proj/norm":4628.550242620596,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/mean":-0.00011920928955078125,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_13/grad/std":4.959220117522524e-05,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/mean":-1.5314435586333275e-07,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/max_abs":0.0004119873046875,"train/train/tensor_act_model_layers_78_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/mean":0.000469207763671875,"train/train/layer_model_layers_35/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_42/grad/frac_near_user_limit":0,"train/train/layer_model_layers_72/act/std":0.7130549069037534,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/std":1.2501673626516407e-05,"train/train/tensor_act_model_layers_19_mlp_down_proj/std":0.038391276411898614,"train/train/tensor_act_model_layers_68_mlp_up_proj/norm":3625.3139025030564,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/norm":6.65625,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/std":5.555011331730279e-05,"train/train/layer_model_layers_40/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_mlp_up_proj/norm":3513.0705484618225,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_76/param/mean":0.0016713730072640406,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/std":8.335852846100642e-05,"train/train/tensor_act_model_layers_76_self_attn/max_abs":1.65625,"train/train/tensor_param_model_embed_tokens_weight/norm":86,"train/train/layer_model_layers_91/grad/max_abs":0.00103759765625,"train/train/tensor_param_model_layers_89_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/mean":6.752088665962219e-09,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/max_abs":0.0002651214599609375,"train/train/tensor_act_model_layers_3_self_attn_o_proj/norm":161.02770829767573,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/mean":-5.476176738739014e-07,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/std":2.728776796482028e-05,"train/train/tensor_act_model_layers_11_self_attn/std":0.052430626464179446,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/norm":0.0008367914739616899,"train/train/tensor_act_model_layers_67_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_mlp_gate_proj_weight/max_abs":0.1708984375,"train/train/tensor_act_model_layers_9_self_attn_v_proj/std":0.28173956279859846,"train/train/tensor_act_model_layers_7_self_attn/std":0.04614781258040305,"train/train/tensor_act_model_layers_19_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_self_attn/max_abs":1.265625,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/mean":2.3888424038887024e-07,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/max_abs":0.265625,"train/train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/max_abs":0.0004749298095703125,"train/train/layer_model_layers_33/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/max_abs":0.1875,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/mean":-2.1219253540039062e-05,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/norm":0.019131771297636737,"train/train/layer__model_layers_48/param/max_abs":1,"train/train/layer_model_layers_18/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/mean":-0.00019550323486328125,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/std":0.04345703125,"train/train/layer_model_layers_33/grad/mean":-7.590351629182812e-08,"train/train/tensor_act_model_layers_40_self_attn_o_proj/std":0.0922858085957331,"train/train/tensor_param_model_layers_16_mlp_gate_proj_weight/std":0.0257568359375,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/max_abs":0.00020122528076171875,"train/train/tensor_act_model_layers_36_mlp_up_proj/norm":2418.8483110689576,"train/train/tensor_act_model_layers_16_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_mlp_gate_proj_weight/max_abs":0.0003986358642578125,"train/train/tensor_act_model_layers_76_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_2/grad/max_abs":0.002685546875,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/max_abs":0.0001983642578125,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_7/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_90/act/norm":20608.259403435357,"train/train/tensor_param_model_layers_69_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_57_mlp_down_proj/max_abs":0.67578125,"train/train/layer__model_layers_75/param/std":0.058238872904820486,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/std":8.524619950172985e-05,"train/train/layer__model_layers_71/param/norm":23.4396249036754,"train/train/tensor_act_model_layers_67_mlp_gate_proj/norm":3497.3284950770144,"train/train/tensor_act_model_layers_1_self_attn_v_proj/std":0.20996099308478927,"train/train/layer_model_layers_58/grad/mean":2.357364863203766e-08,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/norm":3.40625,"train/train/tensor_act_model_layers_64_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/std":4.1343504527571245e-05,"train/train/tensor_act_model_layers_37_mlp/max_abs":0.494140625,"train/train/tensor_act_model_layers_92/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_89/param/mean":0.0015648471397840288,"train/train/layer_model_layers_72/act/max_abs":10.9375,"train/train/tensor_act_model_layers_73_mlp/max_abs":1,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/norm":4.65625,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/max_abs":0.0004787445068359375,"train/train/tensor_act_model_layers_1_mlp_down_proj/max_abs":1.7421875,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_mlp_gate_proj_weight/std":0.03125,"train/train/tensor_act_model_layers_33_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/max_abs":0.00049591064453125,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_self_attn_o_proj/norm":642.2776714730744,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/std":0.0267333984375,"train/train/layer_model_layers_75/grad/mean":-2.203209909745572e-08,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/std":0.00014194394894879113,"train/train/layer_model_layers_80/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/mean":-1.2700911611318588e-07,"train/train/tensor_act_model_layers_8_mlp_down_proj/max_abs":0.6015625,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/max_abs":0.000438690185546875,"train/train/tensor_act_model_layers_6_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_44/grad/mean":-5.692291522453802e-08,"train/train/tensor_act_model_layers_11_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn_q_proj/std":0.946294673808559,"train/train/tensor_act_model_layers_67_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/norm":0.02630820565054255,"train/train/tensor_act_model_layers_91_mlp_down_proj/max_abs":4.09375,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/mean":5.457550287246704e-06,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/norm":7.75,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/max_abs":0.000476837158203125,"train/train/tensor_act_model_layers_56_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/max_abs":0.0005645751953125,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp/norm":238.57130885392115,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_down_proj/mean":0.0002980232238769531,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/norm":0.024205268676273716,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/mean":5.9604644775390625e-06,"train/train/tensor_param_model_layers_90_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_13/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/std":0.0380859375,"train/train/tensor_act_model_layers_88_mlp/norm":1882.608534771745,"train/train/tensor_act_model_layers_54_mlp_down_proj/mean":0.0005865097045898438,"train/train/tensor_act_model_layers_44_mlp_gate_proj/mean":0.002346038818359375,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/max_abs":0.23046875,"train/train/layer_model_layers_18/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_53/act/norm":13917.969108667317,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_71/act/std":0.6990458250320739,"train/train/tensor_act_model_layers_78_self_attn/norm":1086.3418890245548,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/max_abs":0.00015354156494140625,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/std":0.00011263396532407292,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/std":4.771144429064835e-05,"train/train/tensor_param_model_layers_25_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_82_mlp_gate_proj/std":0.523437707059378,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/norm":0.023607669412743705,"train/train/layer_model_layers_45/grad/mean":8.22280828554805e-08,"train/train/tensor_act_model_layers_72_self_attn/max_abs":2.078125,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/std":0.04931640625,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/norm":0.032972996876618796,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_q_proj/max_abs":6.1875,"train/train/layer_model_layers_45/grad/norm":0.03613374577385053,"train/train/layer_model_layers_54/act/std":0.6641156233257518,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/norm":0.01895895790985609,"train/train/layer__model_layers_90/param/max_abs":1,"train/train/tensor_param_model_embed_tokens_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_67/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_input_layernorm/std":1.0000003073364023,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/mean":-2.0157312974333763e-07,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/norm":4.78125,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/mean":-1.7299316823482513e-07,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_mlp_up_proj/std":0.3125000829808304,"train/train/tensor_param_model_layers_13_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/max_abs":0.00055694580078125,"train/train/tensor_param_model_layers_68_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/norm":4.75,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/max_abs":0.201171875,"train/train/tensor_act_model_layers_52/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_k_proj/std":0.8730491343731415,"train/train/tensor_grad_model_layers_55_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84/std":1.9257870790768326,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_mlp/mean":0.0012569427490234375,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/norm":0.017861831849818027,"train/train/tensor_act_model_layers_68_self_attn_q_proj/max_abs":6.40625,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/norm":0.001183454111095735,"train/train/tensor_act_model_layers_65/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/norm":0.0027336025850055673,"train/train/tensor_grad_model_layers_10_mlp_gate_proj_weight/max_abs":0.000476837158203125,"train/train/tensor_act_model_layers_20_mlp/norm":208.38878066925434,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/mean":0.00019359588623046875,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/max_abs":0.000579833984375,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/std":3.8588425282697964e-05,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/max_abs":0.00032806396484375,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/max_abs":0.000705718994140625,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/norm":6.875,"train/train/tensor_act_model_layers_41_mlp_gate_proj/max_abs":2.015625,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/mean":0.0003604888916015625,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/mean":-1.2619420886039734e-07,"train/train/tensor_act_model_layers_19_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/mean":-2.0590960048139095e-09,"train/train/tensor_grad_model_layers_16_mlp_gate_proj_weight/std":3.8818057733151105e-05,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/norm":0.01805062506879954,"train/train/layer_model_layers_78/act/max_abs":11.375,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/max_abs":0.000736236572265625,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/std":3.894464980805803e-05,"train/train/layer_model_layers_81/grad/mean":4.2270421516876696e-07,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/std":7.96388050518756e-05,"train/train/tensor_act_model_layers_90_mlp/mean":0.00630950927734375,"train/train/tensor_act_model_layers_79_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/std":1.0000001415610213,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/mean":-5.316734313964844e-05,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/mean":-3.023305907845497e-07,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/max_abs":0.08203125,"train/train/tensor_act_model_layers_22_mlp_up_proj/norm":2028.936195753212,"train/train/tensor_param_model_layers_84_mlp_gate_proj_weight/std":0.04443359375,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/norm":3.453125,"train/train/tensor_act_model_layers_55_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62/mean":-0.00039440393447875977,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/std":0.040283203125,"train/train/tensor_act_model_layers_81_mlp_gate_proj/max_abs":3.015625,"train/train/tensor_act_model_layers_72_self_attn/mean":-0.0009145736694335938,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/max_abs":0.00049591064453125,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_75_mlp/mean":-0.0020122528076171875,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/std":7.473381864867309e-05,"train/train/tensor_act_model_layers_71_mlp_gate_proj/std":0.4511719668582311,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/std":0.0001433253721527105,"train/train/tensor_act_model_layers_52_mlp_up_proj/std":0.3535161918765586,"train/train/tensor_act_model_layers_28_self_attn_k_proj/norm":5267.8793006499745,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/max_abs":0.0003814697265625,"train/train/layer_model_layers_10/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_7_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_up_proj/max_abs":2.96875,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/norm":5.03125,"train/train/tensor_act_model_layers_35_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/norm":5.3125,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/norm":0.02734742821068586,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/std":0.0400390625,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/std":7.412242139278015e-05,"train/train/tensor_act_model_layers_45_self_attn_v_proj/std":0.32129057840649855,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/std":2.3139621797930178e-05,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/max_abs":0.0004730224609375,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_31/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_q_proj/max_abs":6,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/mean":0.00060272216796875,"train/train/tensor_act_model_layers_11_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_mlp_gate_proj/std":0.31250008298083043,"train/train/tensor_act_model_layers_49/norm":7304.776851414397,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/max_abs":0.00074005126953125,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/norm":9.3125,"train/train/tensor_act_model_layers_8_post_attention_layernorm/norm":5792.605712895219,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/norm":7.46875,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/norm":4.40625,"train/train/tensor_param_model_layers_14_mlp_gate_proj_weight/mean":-0.00012302398681640625,"train/train/tensor_act_model_layers_49_self_attn_o_proj/max_abs":1.84375,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/max_abs":0.0003452301025390625,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/max_abs":0.00168609619140625,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/mean":-6.068497896194458e-06,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/mean":-4.153698682785034e-07,"train/train/tensor_param_model_layers_27_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50/max_abs":9.8125,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_gate_proj_weight/max_abs":0.00150299072265625,"train/train/tensor_param_model_layers_84_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/std":0.036376953125,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/norm":0.006414265706076132,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_41/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_mlp_up_proj/std":0.6113313400241904,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/std":0.0244140625,"train/train/tensor_act_model_layers_36_self_attn_q_proj/std":0.9912124051238221,"train/train/tensor_act_model_layers_54_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/norm":0.006664493384747496,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/norm":0.001674999644162457,"train/train/tensor_act_model_layers_78_mlp_gate_proj/std":0.4921875408599284,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/max_abs":0.001556396484375,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/std":5.3019556251195064e-05,"train/train/tensor_act_model_layers_77/max_abs":11.125,"train/train/tensor_grad_model_layers_65_mlp_gate_proj_weight/std":7.558844260629711e-05,"train/train/tensor_act_model_layers_51_input_layernorm/std":1.0000003474705714,"train/train/tensor_grad_model_layers_85_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/norm":0.016914999145228145,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_42_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/max_abs":0.00069427490234375,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/mean":0.00011777877807617188,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/mean":-0.0001239776611328125,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_9/param/norm":19.530687491899766,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/norm":5.53125,"train/train/tensor_act_model/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_v_proj/max_abs":1.9609375,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/mean":-6.628036499023438e-05,"train/train/tensor_act_model_layers_49_mlp_gate_proj/std":0.3417969819477459,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/max_abs":0.265625,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn/norm":370.8522028811958,"train/train/tensor_act_model_layers_79_self_attn/norm":809.8521964442652,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/max_abs":0.26171875,"train/train/layer__model_layers_68/param/mean":0.00166308898449688,"train/train/tensor_act_model_layers_13_self_attn/norm":328.41662974550667,"train/train/tensor_param_model_layers_40_mlp_gate_proj_weight/mean":0.0002536773681640625,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/norm":0.0170635566174665,"train/train/tensor_act_model_layers_79_post_attention_layernorm/norm":5792.609252931616,"train/train/tensor_act_model_layers_7_post_attention_layernorm/max_abs":5.125,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/max_abs":0.00016307830810546875,"train/train/tensor_act_model_layers_74/mean":0.00904083251953125,"train/train/tensor_act_model_layers_82_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/std":1.9794539104047357e-05,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/mean":9.393692016601562e-05,"train/train/tensor_act_model_layers_28_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_88/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_input_layernorm/norm":5792.6047363341495,"train/train/tensor_act_model_layers_1_self_attn_k_proj/max_abs":3.0625,"train/train/tensor_grad_model_layers_58_mlp_gate_proj_weight/std":6.439785660205116e-05,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/mean":0.0003032684326171875,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/mean":-3.057066351175308e-07,"train/train/tensor_act_model_layers_75_mlp_down_proj/std":0.15625006876651548,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/max_abs":0.125,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/max_abs":0.0002288818359375,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/mean":7.772445678710938e-05,"train/train/tensor_act_model_layers_34_self_attn_v_proj/max_abs":2.71875,"train/train/layer__model_layers_79/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/std":5.162005990814129e-05,"train/train/tensor_act_model_layers_89_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_mlp_gate_proj_weight/mean":4.889443516731262e-09,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4/norm":7507.644090195742,"train/train/tensor_act_model_layers_18_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_gate_proj_weight/norm":0.01679199059018557,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/std":0.0250244140625,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/norm":5.125,"train/train/tensor_param_model_layers_43_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/max_abs":0.154296875,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/std":3.511746416663964e-05,"train/train/tensor_act_model_layers_16_self_attn_k_proj/std":1.072271003084297,"train/train/tensor_act_model_layers_0_mlp/norm":7373.49843335036,"train/train/layer_model_layers_63/act/norm":14031.531846277521,"train/train/tensor_act_model_layers_73_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_mlp_down_proj/norm":357.7946490354836,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/std":4.606962565288027e-05,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_55/grad/norm":0.042473130064966926,"train/train/tensor_param_model_layers_11_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_11_mlp_gate_proj_weight/std":4.1672101881226224e-05,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/norm":0.014989475144037166,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/norm":3.46875,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp_gate_proj/max_abs":2.78125,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/norm":0.00570874512073533,"train/train/tensor_act_model_layers_46/norm":7220.052181310896,"train/train/tensor_act_model_layers_56_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_mlp_gate_proj_weight/max_abs":0.00058746337890625,"train/train/tensor_act_model_layers_80_post_attention_layernorm/max_abs":6.0625,"train/train/tensor_act_model_layers_69_self_attn/max_abs":2.296875,"train/train/tensor_act_model_layers_69_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_mlp_gate_proj_weight/mean":0.00016880035400390625,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/max_abs":0.00138092041015625,"train/train/tensor_act_model_layers_72_mlp_up_proj/norm":3697.893355045637,"train/train/tensor_act_model_layers_39_self_attn/max_abs":1.0078125,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/norm":0.03302876185751433,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/std":4.002084466358365e-05,"train/train/tensor_act_model_layers_45_post_attention_layernorm/mean":-0.011474609375,"train/train/tensor_act_model_layers_27_self_attn/max_abs":0.80078125,"train/train/layer_model_layers_39/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/max_abs":0.000896453857421875,"train/train/tensor_act_model_layers_33_mlp_up_proj/norm":2318.5748853844743,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/norm":0.004870542368085731,"train/train/tensor_act_model_layers_20_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/std":0.00011700125261138775,"train/train/tensor_act_model_layers_41_self_attn_o_proj/norm":512.1543658271897,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/max_abs":0.3671875,"train/train/tensor_act_model_embed_tokens/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/mean":-1.8149148672819138e-07,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/norm":0.0027211923091003246,"train/train/tensor_act_model_layers_18_mlp_down_proj/max_abs":0.470703125,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/mean":1.5157274901866913e-07,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/std":0,"train/train/layer__model_layers_80/param/norm":24.986109422287015,"train/train/tensor_act_model_layers_76_self_attn_k_proj/max_abs":5.875,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/norm":0.01589040065708293,"train/train/tensor_act_model_layers_39_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_o_proj/mean":-0.0003018379211425781,"train/train/tensor_act_model_layers_71_self_attn_v_proj/std":0.4277344767905685,"train/train/tensor_param_model_layers_0_mlp_gate_proj_weight/std":0.04052734375,"train/train/tensor_act_model_layers_27_mlp_gate_proj/norm":2149.393878222965,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/max_abs":0.123046875,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_input_layernorm/mean":-0.0137481689453125,"train/train/tensor_act_model_layers_63_self_attn/mean":0.001739501953125,"train/train/tensor_act_model_layers_85_self_attn_o_proj/norm":1110.1296840109146,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/mean":-5.4191332310438156e-08,"train/train/tensor_act_model_layers_69_self_attn_q_proj/max_abs":5.96875,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/mean":-2.998858690261841e-07,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/max_abs":0.16015625,"train/train/tensor_param_model_layers_84_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_k_proj/norm":4992.524339545137,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/std":7.856978761136805e-05,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/std":1.8073013476774395e-05,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp_up_proj/std":0.3242190195672799,"train/train/tensor_act_model_layers_48_mlp_up_proj/norm":2736.4463990959807,"train/train/tensor_grad_model_layers_76_mlp_gate_proj_weight/std":7.976273574375738e-05,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/std":0.0269775390625,"train/train/tensor_act_model_layers_80/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/norm":0.013569276582382864,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/std":0.00016565454086314984,"train/train/tensor_act_model_layers_78_self_attn_k_proj/norm":5258.212305239763,"train/train/tensor_act_model_layers_8_input_layernorm/norm":5792.60400390909,"train/train/tensor_act_model_layers_51_self_attn_v_proj/std":0.33984378231394796,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp/std":0.1318359485516941,"train/train/tensor_grad_model_layers_87_mlp_gate_proj_weight/norm":0.030916400322619424,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_down_proj/norm":227.8674351137486,"train/train/tensor_act_model_layers_57_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/mean":-0.00021076202392578125,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/max_abs":0.000579833984375,"train/train/tensor_param_model_layers_45_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/max_abs":0.2255859375,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/norm":0.016169018147548685,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/std":0.0260009765625,"train/train/layer_model_layers_57/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_o_proj/norm":896.9917960355883,"train/train/tensor_param_model_layers_91_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/norm":0.011481181758247956,"train/train/tensor_param_model_layers_85_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_70_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/mean":0.0003032684326171875,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/max_abs":0.2158203125,"train/train/tensor_act_model_layers_10_mlp_gate_proj/max_abs":2.09375,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/mean":1.0320218279957771e-07,"train/train/tensor_act_model_layers_79_mlp_up_proj/std":0.4941407946258375,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/norm":4.65625,"train/train/tensor_param_model_layers_85_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_self_attn_o_proj/std":0.21460028722419217,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/max_abs":0.0011749267578125,"train/train/tensor_param_model_layers_6_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_47_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp/mean":0.001064300537109375,"train/train/tensor_param_model_layers_38_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/norm":4.15625,"train/train/tensor_param_model_layers_11_mlp_gate_proj_weight/max_abs":0.11669921875,"train/train/tensor_act_model_layers_91_input_layernorm/max_abs":5.21875,"train/train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6/max_abs":8.4375,"train/train/tensor_grad_model_layers_67_mlp_gate_proj_weight/norm":0.027406756153891612,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_46/std":1.2461003307838423,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_47/act/max_abs":9.625,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/norm":0.018741868163072054,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp_gate_proj/norm":3535.5018795769665,"train/train/tensor_act_model_layers_8_self_attn_v_proj/max_abs":1.6875,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/norm":5.15625,"train/train/tensor_act_model_layers_69_mlp/mean":-0.0005826950073242188,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp/mean":0.0004096031188964844,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/max_abs":0.2021484375,"train/train/tensor_act_model_layers_75_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/norm":0.007999055423889179,"train/train/layer_model_layers_41/act/frac_near_user_limit":0,"train/train/layer_model_layers_92/act/norm":23305.01912647208,"train/train/tensor_act_model_layers_83_input_layernorm/norm":5792.609375001852,"train/train/layer_model_layers_37/grad/std":5.2191789166441636e-05,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/std":4.232500299651491e-05,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/mean":-1.329183578491211e-05,"train/train/tensor_act_model_layers_73_mlp_up_proj/norm":3767.6321187237513,"train/train/tensor_act_model_layers_57_self_attn/std":0.13110669383489126,"train/train/tensor_act_model_layers_85/std":1.9589889476752231,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/max_abs":0.2431640625,"train/train/tensor_param_model_layers_69_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/std":8.784900058483546e-05,"train/train/tensor_act_model_layers_23_post_attention_layernorm/norm":5792.612670900921,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/norm":6.125,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_80/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/std":6.40546515523363e-05,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/std":0.02587890625,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/norm":0.015782338945191546,"train/train/tensor_act_model_layers_91_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/norm":5.375,"train/train/layer_model_layers_44/grad/max_abs":0.00102996826171875,"train/train/tensor_param_model_layers_71_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_59_mlp_gate_proj_weight/max_abs":0.000629425048828125,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/norm":4.53125,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp_down_proj/max_abs":0.55859375,"train/train/tensor_grad_model_embed_tokens_weight/std":0.00013237279019078735,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_act_model_layers_80_self_attn_k_proj/max_abs":5.84375,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/max_abs":0.181640625,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/mean":-2.6868656277656555e-07,"train/train/tensor_act_model_layers_44_post_attention_layernorm/norm":5792.6093750036525,"train/train/layer_model_layers_33/grad/norm":0.03945928295726221,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/norm":2.875,"train/train/layer__model_layers_68/param/max_abs":1,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/mean":-4.194676876068115e-06,"train/train/tensor_act_model_layers_57_mlp/max_abs":0.67578125,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_81/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_down_proj/std":0.09179691421461662,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_mlp_gate_proj_weight/std":0.04052734375,"train/train/tensor_act_model_layers_27_mlp/mean":0.0005645751953125,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/std":5.110504278814446e-05,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/norm":4.21875,"train/train/tensor_act_model_layers_23_self_attn/mean":0.0006527900695800781,"train/train/tensor_act_model_layers_86_self_attn_k_proj/norm":5964.220374748511,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/norm":0.016576748423181374,"train/train/layer__model_layers_66/param/mean":0.0014519654272499025,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_norm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_gate_proj_weight/max_abs":0.0017852783203125,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/norm":0.01447672335045887,"train/train/tensor_act_model_layers_88_mlp_gate_proj/norm":4982.273418092121,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/mean":2.78279185295105e-06,"train/train/tensor_act_model_layers_53_self_attn_k_proj/norm":4640.193412034183,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/mean":-2.5564804673194885e-07,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn_v_proj/max_abs":2.78125,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/std":3.549834919379102e-05,"train/train/tensor_act_model_layers_25_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/max_abs":0.000186920166015625,"train/train/tensor_act_model_layers_77_self_attn_k_proj/max_abs":6,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/max_abs":0.00113677978515625,"train/train/layer_model_layers_72/grad/norm":0.05998866853598599,"train/train/layer_model_layers_81/act/max_abs":11.8125,"train/train/tensor_act_model_layers_41/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/mean":0.00021076202392578125,"train/train/layer_model_layers_9/act/max_abs":8.6875,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/mean":7.182825356721878e-08,"train/train/tensor_act_model_layers_87_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_o_proj/std":0.09576541487301234,"train/train/tensor_act_model_layers_84_self_attn_o_proj/std":0.22070438331005232,"train/train/tensor_act_model_layers_65_input_layernorm/std":1.0000003502699952,"train/train/tensor_act_model_layers_53_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_o_proj/std":0.19751904324949665,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/max_abs":0.11767578125,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/norm":4.25,"train/train/layer__model_layers_77/param/mean":0.001705128019573908,"train/train/tensor_act_model_layers_4/max_abs":8.0625,"train/train/tensor_act_model_layers_50/norm":7325.983641753364,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp_up_proj/max_abs":2.328125,"train/train/tensor_act_model_layers_6_mlp_gate_proj/std":0.2387702065245708,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_mlp_gate_proj_weight/max_abs":0.11083984375,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/mean":7.581710815429688e-05,"train/train/tensor_param_model_layers_38_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_input_layernorm/norm":5792.606567383177,"train/train/tensor_act_model_layers_62_self_attn_k_proj/norm":4892.182868749333,"train/train/tensor_param_model_layers_89_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_57/act/max_abs":9.9375,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/std":0.0625,"train/train/tensor_act_model_layers_83/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/mean":2.55415216088295e-07,"train/train/tensor_act_model_layers_46_self_attn_o_proj/mean":-0.0005030632019042969,"train/train/layer__model_layers_91/param/max_abs":1,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/mean":0.0001983642578125,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/mean":1.3201497495174408e-07,"train/train/tensor_act_model_layers_16_input_layernorm/max_abs":6.09375,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/norm":5.9375,"train/train/tensor_act_model_layers_78_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_mlp/frac_near_dtype_limit":0,"train/train/global/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_mlp_gate_proj_weight/max_abs":0.0006256103515625,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/mean":4.187226295471191e-06,"train/train/tensor_param_model_layers_93_mlp_gate_proj_weight/max_abs":0.27734375,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/mean":-1.0730582289397717e-07,"train/train/tensor_param_model_layers_4_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/norm":0.0042522771095970645,"train/train/tensor_param_model_layers_53_mlp_gate_proj_weight/max_abs":0.154296875,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_o_proj/mean":-0.00040531158447265625,"train/train/tensor_act_model_layers_57_mlp_up_proj/std":0.38867197324281916,"train/train/tensor_act_model_layers_72_mlp_up_proj/std":0.4511720718759049,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp_down_proj/std":0.04925549169468252,"train/train/layer_model_layers_64/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp_down_proj/max_abs":8.4375,"train/train/tensor_param_model_layers_20_mlp_gate_proj_weight/mean":-0.0001983642578125,"train/train/layer_model_layers_33/act/std":0.6328781204492223,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/max_abs":0.0003910064697265625,"train/train/tensor_act_model_layers_64_mlp/max_abs":0.83203125,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_10_self_attn_o_proj/max_abs":1.171875,"train/train/tensor_act_model_layers_15_self_attn_q_proj/mean":0.0125274658203125,"train/train/tensor_act_model_layers_9_self_attn_k_proj/norm":4868.7699036020895,"train/train/tensor_act_model_layers_66_self_attn/mean":0.003326416015625,"train/train/tensor_act_model_layers_13_self_attn_k_proj/max_abs":5.1875,"train/train/tensor_act_model_layers_51_self_attn_v_proj/norm":1968.3816903617046,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/norm":0.016638434359927026,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/std":7.701976312374879e-05,"train/train/tensor_act_model_layers_19_self_attn/mean":-0.00019156932830810547,"train/train/tensor_param_model_layers_91_mlp_gate_proj_weight/std":0.05419921875,"train/train/tensor_act_model_layers_75_self_attn/norm":956.0093340802019,"train/train/tensor_act_model_layers_59_self_attn/norm":810.8501244792732,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/max_abs":0.171875,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/std":4.852690225604455e-05,"train/train/tensor_act_model_layers_69_mlp_down_proj/max_abs":0.94140625,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/norm":0.01335920707418631,"train/train/layer_model_layers_42/grad/mean":5.027107813057029e-09,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/norm":5.3125,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/mean":-3.546476364135742e-06,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/max_abs":0.138671875,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/mean":9.080395102500916e-09,"train/train/tensor_act_model_layers_50_mlp_up_proj/norm":2861.9753748167946,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/norm":0.06923356356238904,"train/train/layer_model_layers_52/grad/max_abs":0.00136566162109375,"train/train/tensor_act_model_layers_16_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/max_abs":0.2578125,"train/train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/max_abs":0.11669921875,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/norm":0.0024007632902872016,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/max_abs":0.2119140625,"train/train/layer_model_layers_70/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/norm":3.546875,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/max_abs":0.000713348388671875,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_mlp_gate_proj_weight/max_abs":0.16796875,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp/mean":0.0008382797241210938,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_post_attention_layernorm/norm":5792.599609380525,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/max_abs":0.000514984130859375,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/norm":6.75,"train/train/tensor_param_model_layers_31_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp_up_proj/norm":6113.808234284625,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_v_proj/norm":1658.3585594068672,"train/train/tensor_act_model_layers_71_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_mlp/max_abs":1.703125,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/norm":0.006584976341097162,"train/train/tensor_act_model_layers_37_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_22/act/mean":-0.0016806977135794504,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/max_abs":0.2373046875,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/std":5.632215738752398e-05,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/norm":6.65625,"train/train/tensor_act_model_layers_20_input_layernorm/mean":-0.024749755859375,"train/train/tensor_act_model_layers_73_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/mean":2.0372681319713593e-07,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_o_proj/max_abs":2.296875,"train/train/tensor_act_model_layers_36_self_attn_k_proj/mean":-0.029571533203125,"train/train/tensor_act_model_layers_51_self_attn_q_proj/mean":0.0015363693237304688,"train/train/tensor_act_model_layers_69_self_attn_q_proj/norm":5311.139453883333,"train/train/layer_model_layers_6/act/mean":-0.012794664927891322,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/std":0.03857421875,"train/train/tensor_act_/max_abs":2.0224661827087402,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/std":1.8098815232263835e-05,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_input_layernorm/norm":5792.61633300829,"train/train/layer_model_layers_82/act/max_abs":11.375,"train/train/tensor_act_model_layers_29_self_attn/max_abs":0.8984375,"train/train/tensor_act_model_layers_42_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_52/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65/max_abs":10.1875,"train/train/layer__model_layers_54/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/max_abs":0.26171875,"train/train/tensor_grad_model_layers_36_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_mlp/norm":1034.3043403417032,"train/train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/norm":7.84375,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/norm":0.0006190044263800766,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/mean":-4.1763996705412865e-08,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/mean":0.0002498626708984375,"train/train/tensor_act_model_layers_43_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_54/grad/norm":0.051525466877081365,"train/train/tensor_act_model_layers_78_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_input_layernorm/std":1.0000012754454526,"train/train/tensor_act_model_layers_65_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/mean":-1.3396143913269043e-05,"train/train/tensor_act_model_layers_88_self_attn_o_proj/norm":1117.325149057583,"train/train/tensor_act_model_layers_67/mean":0.00655364990234375,"train/train/layer__model_layers_84/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/norm":8.25,"train/train/tensor_act_model_layers_51_self_attn_k_proj/max_abs":5.28125,"train/train/tensor_act_model_layers_73_input_layernorm/norm":5792.609008789477,"train/train/layer_model_layers_45/act/max_abs":9.4375,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_gate_proj/norm":1979.8670349731694,"train/train/tensor_act_model_layers_92_self_attn/std":0.25391006229019114,"train/train/tensor_act_model_layers_21_self_attn_k_proj/mean":0.04150390625,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/norm":0.019420116869774167,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/mean":-2.973247319459915e-07,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/mean":-3.269314765930176e-05,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/max_abs":0.1279296875,"train/train/tensor_act_model_layers_79_self_attn_o_proj/std":0.13989494957724355,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/max_abs":0.00038909912109375,"train/train/tensor_act_model_layers_40_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/mean":-0.000621795654296875,"train/train/tensor_act_model_layers_93_self_attn_v_proj/mean":0.01702880859375,"train/train/layer__model_layers_74/param/max_abs":1,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/std":4.741956130506224e-05,"train/train/tensor_act_model_layers_13_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/norm":0.000815359750763768,"train/train/tensor_act_model_layers_72_mlp_down_proj/std":0.14453125519464036,"train/train/layer_model_layers_42/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/max_abs":0.000579833984375,"train/train/layer__model_layers_60/param/norm":22.493054483551138,"train/train/tensor_act_model_layers_38_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/std":0.0595703125,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/max_abs":0.0004119873046875,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_71_mlp_gate_proj_weight/norm":7.09375,"train/train/tensor_act_model_layers_9_self_attn_q_proj/max_abs":6.65625,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_q_proj/std":0.8486441063363261,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_o_proj/norm":378.32849028236365,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/std":0.0001053239603865211,"train/train/tensor_act_model_layers_26_self_attn/mean":-0.00015032291412353516,"train/train/tensor_act_model_layers_20_self_attn_q_proj/mean":1.3828277587890625e-05,"train/train/tensor_act_model_layers_19_mlp_gate_proj/max_abs":1.71875,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/max_abs":0.000919342041015625,"train/train/layer_model_layers_67/act/std":0.6794430895411583,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/std":8.44337347752679e-05,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/mean":0.00029754638671875,"train/train/tensor_param_model_layers_40_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/std":0.04833984375,"train/train/tensor_act_model_layers_52_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp/std":0.18139703397983267,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn_v_proj/max_abs":2.796875,"train/train/tensor_act_model_layers_83/std":1.8808648517335547,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn_q_proj/norm":6031.175224065401,"train/train/tensor_act_model_layers_52_self_attn/mean":-0.00031566619873046875,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/std":4.121250277730875e-05,"train/train/tensor_act_model_layers_17_self_attn/mean":0.00018143653869628906,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/norm":4.59375,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn/std":0.02780239516545149,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/std":2.572269660409138e-05,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/std":0.0206298828125,"train/train/tensor_act_model_layers_25_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp_up_proj/std":0.554687709908878,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/max_abs":0.1865234375,"train/train/tensor_param_model_layers_42_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_76/act/norm":16262.80011721524,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/norm":0.004123262494627957,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/std":0.035888671875,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_mlp_gate_proj_weight/max_abs":0.1630859375,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_23_mlp_up_proj/mean":-0.001678466796875,"train/train/tensor_act_model_layers_38_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/std":4.5198049710677045e-05,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/max_abs":0.1318359375,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/mean":-1.6088597476482391e-07,"train/train/tensor_grad_model_layers_64_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_mlp_gate_proj_weight/norm":0.021278172078544853,"train/train/tensor_act_model_layers_52_self_attn_o_proj/max_abs":0.59765625,"train/train/tensor_act_model_layers_20_self_attn_v_proj/norm":1634.7973782615636,"train/train/tensor_act_model_layers_50_post_attention_layernorm/mean":-0.00733184814453125,"train/train/tensor_grad_model_layers_43_mlp_gate_proj_weight/max_abs":0.000545501708984375,"train/train/tensor_act_model_layers_37_mlp_up_proj/max_abs":2.21875,"train/train/tensor_act_model_layers_66/norm":8086.375482999212,"train/train/tensor_act_model_layers_52_mlp_down_proj/mean":0.0007600784301757812,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/norm":0.018502614081476514,"train/train/tensor_act_model_layers_88_post_attention_layernorm/max_abs":5.65625,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_up_proj/norm":2601.5708556022983,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/mean":8.968636393547058e-07,"train/train/tensor_param_model_layers_90_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/std":0.031982421875,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_14_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_q_proj/norm":5494.5581187878415,"train/train/layer_model_layers_47/grad/mean":-1.4034348326428633e-07,"train/train/tensor_act_model_layers_3_input_layernorm/std":1.0000001317821356,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/norm":0.000701720275805313,"train/train/tensor_act_model_layers_71_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/mean":-0.00017070770263671875,"train/train/layer_model_layers_40/grad/norm":0.038810872225222344,"train/train/tensor_act_model_layers_63_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/mean":0.00030517578125,"train/train/layer_model_layers_9/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn_q_proj/mean":0.00565338134765625,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/std":0.0269775390625,"train/train/tensor_act_model_layers_91_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_mlp_gate_proj_weight/mean":4.982948303222656e-05,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_input_layernorm/mean":0.007965087890625,"train/train/tensor_act_model_layers_32_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/std":0.00010055529921918619,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/std":7.117730708186716e-05,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/max_abs":0.00058746337890625,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/max_abs":0.000858306884765625,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/norm":0.0019571478957878643,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_40/mean":-0.01727294921875,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/mean":0.0001697540283203125,"train/train/tensor_act_model_layers_29_self_attn/std":0.08105543700923366,"train/train/tensor_param_model_layers_17_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/norm":0.013038615164788731,"train/train/tensor_act_model_layers_42_input_layernorm/mean":-0.0133056640625,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/norm":4.75,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/mean":0.000492095947265625,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/std":0.052001953125,"train/train/layer_model_layers_83/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn_v_proj/norm":2160.407192646497,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/norm":5.1875,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp_up_proj/mean":0.01123046875,"train/train/tensor_param_model_layers_12_mlp_gate_proj_weight/max_abs":0.111328125,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/norm":2.734375,"train/train/tensor_act_model_layers_63_mlp/mean":5.7816505432128906e-05,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/max_abs":0.000385284423828125,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/norm":3.453125,"train/train/tensor_act_model_layers_76_post_attention_layernorm/std":1.0000019990291578,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/norm":0.029084837030725067,"train/train/tensor_act_model_layers_62_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn/norm":557.4589458849216,"train/train/tensor_act_model_layers_88/norm":12472.027412035317,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/norm":0.004517864360061785,"train/train/tensor_act_model_layers_20_mlp_down_proj/mean":0.001190185546875,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/norm":5.71875,"train/train/tensor_act_model_layers_64_mlp_up_proj/max_abs":2.390625,"train/train/layer_model_layers_78/act/mean":-0.001902001244681222,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/std":8.854106065840009e-05,"train/train/tensor_act_model_layers_71_self_attn_q_proj/std":0.9443386752415355,"train/train/tensor_act_model_layers_54_self_attn_o_proj/std":0.13940672256917103,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/std":0.0262451171875,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/norm":0.01630312890872978,"train/train/layer_model_layers_92/grad/norm":0.0703419360493495,"train/train/tensor_act_model_layers_10_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_o_proj/max_abs":2.296875,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/std":0.0242919921875,"train/train/tensor_act_model_layers_79_mlp_up_proj/mean":-0.01385498046875,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/max_abs":0.125,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87/std":2.0898564146221585,"train/train/layer_model_layers_57/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/norm":0.02266533768020891,"train/train/tensor_act_model_layers_43_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/std":5.5713462022419286e-05,"train/train/tensor_act_model_layers_61_self_attn_q_proj/max_abs":9.5625,"train/train/layer_model_layers_43/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/std":0.00011330200308040927,"train/train/tensor_act_model_layers_32/norm":7222.617481434823,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_mlp_gate_proj_weight/std":0.031005859375,"train/train/tensor_act_model_layers_61_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/norm":0.009087425598800195,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/mean":0.0001277923583984375,"train/train/tensor_param_model_layers_45_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/max_abs":0.000946044921875,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_48_self_attn_k_proj/max_abs":4.71875,"train/train/tensor_act_model_layers_32_mlp_up_proj/std":0.27929692118187294,"train/train/tensor_act_model_layers_67_self_attn_o_proj/max_abs":1.8359375,"train/train/tensor_act_model_layers_52_post_attention_layernorm/norm":5792.610473634936,"train/train/layer_model_layers_86/grad/norm":0.07338569410454175,"train/train/tensor_act_model_layers_71_mlp_down_proj/max_abs":1.1953125,"train/train/tensor_act_model_layers_55_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_o_proj/norm":405.60985724028205,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_56_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_post_attention_layernorm/mean":0.008636474609375,"train/train/tensor_act_model_layers_50_self_attn_k_proj/max_abs":5.4375,"train/train/tensor_act_model_layers_93_self_attn_q_proj/norm":6544.575586483154,"train/train/tensor_act_model_layers_66_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_k_proj/mean":-0.007537841796875,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/max_abs":0.1259765625,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_up_proj/mean":-0.00388336181640625,"train/train/tensor_act_model_layers_21_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_q_proj/max_abs":5,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/mean":-6.908085197210312e-07,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/mean":-0.0005950927734375,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/norm":0.005781550947295007,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/mean":3.9173755794763565e-08,"train/train/tensor_act_model_layers_28/mean":-0.0198974609375,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/max_abs":0.00010824203491210938,"train/train/layer__model_layers_76/param/max_abs":1,"train/train/tensor_act_model_layers_23_self_attn_o_proj/norm":283.9690646290212,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/norm":0.02575958381928235,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp/mean":-0.002292633056640625,"train/train/tensor_param_model_layers_12_mlp_gate_proj_weight/std":0.0255126953125,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/norm":6.75,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_92/param/norm":26.808693640076907,"train/train/tensor_act_model_layers_14_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/mean":0.00016689300537109375,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/norm":0.0025849014688993153,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/mean":0.0001926422119140625,"train/train/tensor_act_model_layers_31/max_abs":9,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_5_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_93/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_input_layernorm_weight/std":0,"train/train/layer_model_layers_11/grad/std":5.2812503331393385e-05,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_33/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/norm":0.0012810569041983653,"train/train/tensor_param_model_layers_63_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp_gate_proj/std":0.3457031318191754,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/max_abs":0.2294921875,"train/train/tensor_act_model_layers_64_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_gate_proj_weight/mean":5.168840289115906e-07,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/max_abs":0.251953125,"train/train/tensor_act_model_layers_45_self_attn_q_proj/std":0.7968798805536147,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/std":0.04638671875,"train/train/tensor_act_model_layers_4_self_attn_o_proj/std":0.13794017358766877,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/std":0.019287109375,"train/train/tensor_act_model_layers_22_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_o_proj/norm":554.7975450313287,"train/train/tensor_act_model_layers_4_input_layernorm/max_abs":4.90625,"train/train/tensor_act_model_layers_17_mlp_gate_proj/norm":2045.6683775103656,"train/train/tensor_act_model_layers_58_mlp_up_proj/mean":-0.003459930419921875,"train/train/tensor_act_model_layers_69_self_attn_v_proj/mean":-0.00464630126953125,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/std":0.031982421875,"train/train/tensor_param_model_layers_0_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/mean":3.0535738915205e-07,"train/train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/norm":7.40625,"train/train/tensor_act_model_layers_5_mlp/mean":-0.0003490447998046875,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_v_proj/std":0.42480589349893405,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/norm":0.01984027531757754,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_70/act/norm":15337.129062530681,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/mean":1.825392246246338e-05,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_52_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_79_self_attn/max_abs":1.828125,"train/train/tensor_act_model_layers_69_post_attention_layernorm/std":1.0000018005941615,"train/train/tensor_act_model_layers_76_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_mlp_gate_proj_weight/std":0.026123046875,"train/train/tensor_act_model_layers_14/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5/max_abs":8.125,"train/train/tensor_act_model_layers_11_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/max_abs":0.0002307891845703125,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/std":2.7781762454431192e-05,"train/train/tensor_act_model_layers_49_self_attn_q_proj/mean":0.03704833984375,"train/train/tensor_act_model_layers_70_post_attention_layernorm/max_abs":5.71875,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_62/grad/std":6.234538480791214e-05,"train/train/tensor_grad_model_layers_55_mlp_gate_proj_weight/norm":0.021888827313405395,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/max_abs":0.00016117095947265625,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/max_abs":0.154296875,"train/train/layer_model_layers_65/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_5_input_layernorm/norm":5792.608398438354,"train/train/tensor_act_model_layers_79_input_layernorm/mean":0.0059967041015625,"train/train/tensor_param_model_layers_72_mlp_gate_proj_weight/norm":7.125,"train/train/tensor_act_model_layers_32_self_attn/std":0.08667385721905517,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/std":0.0264892578125,"train/train/tensor_act_model_layers_42_input_layernorm/max_abs":6.59375,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/max_abs":0.23046875,"train/train/tensor_act_model_layers_74_self_attn_v_proj/max_abs":3,"train/train/tensor_act_model_layers_36_mlp/norm":292.7850369317611,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/std":0.038330078125,"train/train/tensor_param_model_layers_55_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/norm":0.015536834407461654,"train/train/tensor_act_model_layers_25_self_attn_v_proj/mean":0.0021533966064453125,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp_up_proj/mean":-0.0018253326416015625,"train/train/layer_model_layers_78/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/std":0.046875,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_mlp_down_proj/mean":-6.955862045288086e-05,"train/train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_input_layernorm/std":1.0000005421000773,"train/train/tensor_act_model_layers_92_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_o_proj/std":0.06494519366741308,"train/train/tensor_act_model_layers_85_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_mlp_gate_proj_weight/std":4.084651388030473e-05,"train/train/tensor_grad_model_layers_68_mlp_gate_proj_weight/std":7.011169265316136e-05,"train/train/tensor_act_model_layers_9_self_attn_q_proj/mean":0.0160369873046875,"train/train/tensor_act_model_layers_55_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_k_proj/mean":-0.022186279296875,"train/train/tensor_act_model_layers_31_input_layernorm/norm":5792.602294927375,"train/train/tensor_act_model_layers_43/norm":7191.961533754848,"train/train/layer__model_layers_86/param/std":0.06281494648013719,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/norm":6.65625,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/max_abs":0.0003795623779296875,"train/train/tensor_act_model_layers_49_self_attn_v_proj/norm":2235.9970608246394,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/std":0.0400390625,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/mean":-1.2683449313044548e-07,"train/train/layer__model_layers_78/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/std":0.037109375,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/std":4.665727278595557e-05,"train/train/layer_model_layers_43/grad/max_abs":0.001556396484375,"train/train/layer_model_layers_12/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/std":0.0001117230140411288,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/max_abs":0.00188446044921875,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_mlp_gate_proj_weight/max_abs":0.1171875,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/norm":6.15625,"train/train/tensor_act_model_layers_35/std":1.2343761456912796,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/max_abs":0.2451171875,"train/train/tensor_param_model_layers_35_mlp_gate_proj_weight/norm":5.25,"train/train/tensor_act_model_layers_37_self_attn_o_proj/norm":653.9716916525397,"train/train/tensor_act_model_layers_68_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_gate_proj_weight/std":0.030517578125,"train/train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_v_proj/std":0.44726564699385307,"train/train/tensor_act_model_layers_92_mlp_gate_proj/max_abs":4.65625,"train/train/tensor_act_model_layers_90_mlp/max_abs":3.078125,"train/train/tensor_act_model_layers_0_mlp_up_proj/mean":-0.01483154296875,"train/train/tensor_act_model_layers_64_post_attention_layernorm/norm":5792.6114501959955,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_73/act/std":0.7090290840423259,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_o_proj/norm":370.8522028811958,"train/train/tensor_act_model_layers_54_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/max_abs":0.0003299713134765625,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/norm":4.875,"train/train/tensor_act_model_layers_28_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/max_abs":0.1826171875,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/norm":4.3125,"train/train/tensor_act_model_layers_18_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_34/param/mean":0.0015157381197591655,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/mean":-3.170967102050781e-05,"train/train/tensor_act_model_layers_2_input_layernorm/norm":5792.603759766052,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_28_input_layernorm/std":1.0000013397066263,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/mean":0.000377655029296875,"train/train/tensor_act_model_layers_52_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_24/act/std":0.6141626011905994,"train/train/tensor_act_model_layers_44_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/std":6.207267661205977e-05,"train/train/tensor_act_model_layers_15_post_attention_layernorm/std":1.000000921543263,"train/train/tensor_param_model_layers_46_mlp_gate_proj_weight/max_abs":0.16796875,"train/train/tensor_param_model_layers_24_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_mlp_gate_proj_weight/norm":0.014792843159600047,"train/train/tensor_grad_model_layers_59_mlp_gate_proj_weight/std":6.736983456469117e-05,"train/train/tensor_act_model_layers_57_input_layernorm/norm":5792.606201176531,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/mean":5.841255187988281e-06,"train/train/tensor_act_model_layers_10_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/norm":0.005629293836106142,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/norm":0.014076842834987374,"train/train/tensor_act_model_layers_66_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp/mean":-0.000762939453125,"train/train/tensor_param_model_layers_29_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_70_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp_up_proj/max_abs":1.8828125,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp_gate_proj/max_abs":2.125,"train/train/tensor_act_model_layers_62/max_abs":9.75,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/mean":1.1166557669639587e-06,"train/train/tensor_act_model_layers_13_self_attn_v_proj/mean":0.00022017955780029297,"train/train/tensor_act_model_layers_30_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_9/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/std":1.538591451873331e-05,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/max_abs":0.000354766845703125,"train/train/tensor_act_model_layers_72_self_attn_v_proj/frac_near_dtype_limit":0,"eval/runtime":18.7782,"train/train/tensor_act_model_layers_86_mlp_gate_proj/max_abs":3.4375,"train/train/tensor_act_model_layers_8_post_attention_layernorm/std":1.0000004675238232,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/mean":-1.0073272278532386e-06,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/std":0.034423828125,"train/train/tensor_act_model_layers_73_self_attn_k_proj/std":0.8291129778692832,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/max_abs":0.00013065338134765625,"train/train/tensor_act_model_layers_32_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_21/param/mean":0.001534543803627145,"train/train/tensor_param_model_layers_78_mlp_gate_proj_weight/std":0.041748046875,"train/train/tensor_param_model_layers_79_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/norm":0.002755880510531673,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/std":2.201972764685179e-05,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/mean":-6.146728992462158e-08,"train/train/tensor_act_model_layers_74_self_attn_q_proj/norm":6320.796229507938,"train/train/tensor_act_model_layers_89_mlp_down_proj/std":0.3701185554063771,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/mean":0.0001659393310546875,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/max_abs":0.0010833740234375,"train/train/tensor_act_model_layers_68_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/std":8.697137404829894e-05,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/max_abs":0.224609375,"train/train/tensor_act_model_layers_82_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/std":5.3621131367288284e-05,"train/train/tensor_act_model_layers_93_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/mean":-0.000316619873046875,"train/train/tensor_act_model_layers_16_self_attn_v_proj/mean":0.0012941360473632812,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/norm":0.004633272944208061,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/norm":3.9375,"train/train/tensor_act_model_layers_83_mlp_gate_proj/std":0.5351564779088612,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/norm":0.020792752717906243,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/mean":1.3518729247152805e-07,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/mean":-0.0006256103515625,"train/train/tensor_param_model_layers_56_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_k_proj/max_abs":5.625,"train/train/tensor_act_model_layers_10_mlp_down_proj/norm":229.24278567897133,"train/train/tensor_act_model_layers_36_mlp_up_proj/max_abs":1.890625,"train/train/tensor_act_model_layers_58_post_attention_layernorm/mean":0.00018215179443359375,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/mean":-3.4319236874580383e-07,"train/train/tensor_act_model_layers_14_input_layernorm/max_abs":5.9375,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/mean":-0.0001850128173828125,"train/train/tensor_act_model_layers_30_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_down_proj/mean":0.0005617141723632812,"train/train/tensor_act_model_layers_57_self_attn_q_proj/norm":5416.46214163135,"train/train/tensor_act_model_layers_62_mlp_up_proj/std":0.40234392186966716,"train/train/tensor_act_model_layers_5/mean":-0.031097412109375,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/max_abs":0.000774383544921875,"train/train/tensor_act_model_layers_60_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp_gate_proj/mean":-0.00136566162109375,"train/train/tensor_act_model_layers_2_input_layernorm/max_abs":4.84375,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/max_abs":0.000385284423828125,"train/train/tensor_act_model_layers_71_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_gate_proj_weight/max_abs":0.0003795623779296875,"train/train/tensor_act_model_layers_91_mlp_up_proj/norm":5766.123973172178,"train/train/tensor_act_model_layers_76_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_mlp_gate_proj_weight/mean":0.0004253387451171875,"train/train/layer__model_layers_47/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_k_proj/max_abs":6,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/std":0.03759765625,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/norm":4.71875,"train/train/tensor_act_model_layers_4_mlp_gate_proj/std":0.24707033653032992,"train/train/tensor_grad_model_layers_78_mlp_gate_proj_weight/std":8.024434595021437e-05,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/mean":0.00012302398681640625,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/norm":0.0012814600960527464,"train/train/tensor_act_model_layers_22/mean":-0.025665283203125,"train/train/tensor_act_model_layers_8_self_attn_q_proj/std":0.9765625839233363,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_input_layernorm/std":1.000001550847335,"train/train/layer_model_layers_72/grad/std":7.407946360982893e-05,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/std":5.279993245715793e-05,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/mean":-1.3568205758929253e-07,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/max_abs":0.00011301040649414062,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_q_proj/std":0.8359536209512818,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/norm":6.59375,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/max_abs":0.00115203857421875,"train/train/tensor_act_model_layers_85_mlp_down_proj/mean":0.00223541259765625,"train/train/tensor_param_model_layers_21_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55/std":1.2890630597586121,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/norm":5.28125,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/mean":0.000446319580078125,"train/train/tensor_act_model_layers_35_self_attn_v_proj/norm":1770.3651462748066,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_input_layernorm/std":1.0000011384247842,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_14/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30/max_abs":9.0625,"train/train/tensor_act_model_layers_35_self_attn_q_proj/mean":-0.00449371337890625,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_input_layernorm/std":1.0000008847560546,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/max_abs":0.2109375,"train/train/tensor_param_model_layers_28_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/norm":0.014849334218532647,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/max_abs":0.0003108978271484375,"train/train/tensor_param_model_layers_63_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_58_mlp_down_proj/mean":0.0010890960693359375,"train/train/tensor_act_model_layers_26/mean":-0.020843505859375,"train/train/tensor_act_model_layers_53_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/mean":7.753260433673859e-08,"train/train/tensor_act_model_layers_37_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/mean":0.0001621246337890625,"train/train/tensor_act_model_layers_69_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_mlp_gate_proj_weight/norm":6.46875,"train/train/tensor_act_model_layers_0_mlp_gate_proj/max_abs":6.4375,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/norm":6.8125,"train/train/layer_model_layers_4/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/max_abs":0.000278472900390625,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/std":0.02392578125,"train/train/tensor_act_model_layers_67_self_attn_v_proj/mean":0.000827789306640625,"train/train/tensor_act_model_layers_61_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp_down_proj/mean":-0.000762939453125,"train/train/tensor_act_model_layers_40_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_k_proj/max_abs":4.4375,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_up_proj/mean":-0.003116607666015625,"train/train/tensor_act_model_layers_49_self_attn_v_proj/max_abs":2.75,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/mean":-5.245208740234375e-05,"train/train/tensor_act_model_layers_75_self_attn_o_proj/mean":-0.0022735595703125,"train/train/tensor_act_model_layers_2_self_attn_q_proj/max_abs":5.53125,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/max_abs":0.00154876708984375,"train/train/tensor_param_model_layers_51_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/mean":-2.8014183044433594e-05,"train/train/tensor_act_model_layers_45_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_mlp_gate_proj/std":0.6093750290381595,"train/train/layer_model_layers_3/act/max_abs":8,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/mean":2.0649167709052563e-08,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/max_abs":0.00118255615234375,"train/train/tensor_act_model_layers_72_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_22/param/max_abs":1,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/std":3.464562299967347e-05,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_64_self_attn_o_proj/mean":0.00013637542724609375,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/std":0.032470703125,"train/train/tensor_act_model_layers_2_self_attn_v_proj/std":0.23974665107848084,"train/train/tensor_act_model_layers_52_mlp/frac_near_dtype_limit":0,"train/train/global/act/std":0.8384323601350128,"train/train/layer_model_layers_50/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/max_abs":0.0007171630859375,"train/train/tensor_act_model_layers_54_mlp_up_proj/std":0.36718765765108113,"train/train/tensor_act_model_layers_8_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/mean":1.469743438065052e-08,"train/train/tensor_act_model_layers_80/max_abs":11.375,"train/train/layer__model_layers_57/param/mean":0.0014973966268221042,"train/train/tensor_act_model_layers_36_mlp_down_proj/mean":0.0003476142883300781,"train/train/tensor_grad_model_layers_77_mlp_gate_proj_weight/norm":0.02881350717832215,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/mean":-7.949769496917725e-06,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/std":0.00038749729012981253,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_17/act/norm":14090.0599157993,"train/train/tensor_act_model_layers_25/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_k_proj/norm":5876.871533452266,"train/train/layer__model_layers_42/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/max_abs":0.0015411376953125,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp_down_proj/std":0.10009766871609262,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/max_abs":0.2275390625,"train/train/tensor_act_model_layers_5_mlp_down_proj/std":0.047120097590244485,"train/train/tensor_param_model_layers_44_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp/max_abs":0.86328125,"train/train/tensor_act_model_layers_50_post_attention_layernorm/norm":5792.61083984553,"train/train/tensor_act_model_layers_90_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_gate_proj_weight/std":0.04541015625,"train/train/layer_model_layers_40/act/std":0.6324318438301906,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/max_abs":0.1416015625,"train/train/tensor_act_model_layers_70_self_attn/norm":1053.7093850396186,"train/train/tensor_act_model_layers_63_mlp_down_proj/mean":5.7816505432128906e-05,"train/train/tensor_act_model_layers_21_mlp_gate_proj/max_abs":2.3125,"train/train/tensor_act_model_layers_59_self_attn_q_proj/norm":5894.81207186363,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/std":0.05126953125,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/std":5.229433458423055e-05,"train/train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_up_proj/mean":0.0018978118896484375,"train/train/tensor_act_model_layers_88_input_layernorm/std":1.0000008018684152,"train/train/tensor_act_model_layers_13_mlp_up_proj/mean":0.00310516357421875,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/mean":8.157803677022457e-08,"train/train/tensor_act_model_layers_17_mlp_gate_proj/max_abs":2.4375,"train/train/tensor_act_model_layers_87_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70/norm":8501.116456075575,"train/train/tensor_grad_model_layers_18_mlp_gate_proj_weight/std":4.115108132200652e-05,"train/train/tensor_act_model_layers_2_self_attn_k_proj/mean":0.0936279296875,"train/train/tensor_act_model_layers_86_self_attn_v_proj/std":0.4809577891058203,"train/train/tensor_act_model_layers_76_mlp_down_proj/std":0.1601562649011605,"train/train/layer_model_layers_18/act/std":0.6185457145110799,"train/train/tensor_act_model_layers_58_self_attn_k_proj/mean":0.0638427734375,"train/train/tensor_param_model_layers_23_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/mean":-0.0003910064697265625,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/std":4.585309014174589e-05,"train/train/tensor_act_model_layers_49_mlp/std":0.0683593814926485,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/std":2.86200649347813e-05,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/std":0.044677734375,"train/train/tensor_act_model_layers_3/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/norm":0.024661073423212974,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/std":4.0498573711677657e-05,"train/train/tensor_act_model_layers_7/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/mean":-0.0003452301025390625,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/max_abs":0.0004596710205078125,"train/train/tensor_act_model_layers_8_mlp_up_proj/mean":0.0030364990234375,"train/train/tensor_act_model_layers_22_self_attn_o_proj/max_abs":0.7109375,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_o_proj/norm":714.5199832580456,"train/train/tensor_act_model_layers_37_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_mlp_gate_proj_weight/norm":0.018072385056415487,"train/train/tensor_act_model_layers_2_mlp/norm":1066.3355648733557,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_77_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_18/param/mean":0.0016633032264649962,"train/train/tensor_act_model_layers_89_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/norm":6.59375,"train/train/tensor_act_model_layers_84_mlp_up_proj/mean":-0.0160369873046875,"train/train/tensor_param_model_layers_32_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_mlp_up_proj/norm":2282.532502356498,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/mean":-1.210719347000122e-06,"train/train/tensor_param_model_layers_16_mlp_gate_proj_weight/max_abs":0.12158203125,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/max_abs":0.1123046875,"train/train/tensor_param_model_layers_1_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_38_mlp_down_proj/norm":320.6065601583395,"train/train/tensor_act_model_layers_55_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/norm":4.53125,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/norm":0.018930149892968234,"train/train/tensor_act_model_layers_14_input_layernorm/norm":5792.603027344286,"train/train/tensor_act_model_layers_40_self_attn_o_proj/mean":0.0002892017364501953,"train/train/tensor_grad_model_layers_89_mlp_gate_proj_weight/mean":4.4796615839004517e-07,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/std":7.041973712807015e-05,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/std":0.041259765625,"train/train/tensor_param_model_layers_83_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_gate_proj/mean":-0.0010700225830078125,"train/train/tensor_param_model_layers_64_mlp_gate_proj_weight/max_abs":0.16015625,"train/train/tensor_grad_model_layers_45_mlp_gate_proj_weight/norm":0.017952015589340942,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/max_abs":0.00102996826171875,"train/train/tensor_act_model_layers_47_mlp_gate_proj/std":0.3339843823198686,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/std":0.042236328125,"train/train/tensor_act_model_layers_34_self_attn_o_proj/max_abs":1.0546875,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/std":0.04150390625,"train/train/tensor_act_model_layers_83_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/std":2.3068089792647327e-05,"train/train/tensor_param_model_layers_64_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/mean":-1.712469384074211e-07,"train/train/tensor_act_model_layers_3_self_attn/mean":-0.0005626678466796875,"train/train/tensor_act_model_layers_51_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/std":4.727102632311133e-05,"train/train/layer__model_layers_7/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_10_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_5_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/norm":3.5,"train/train/tensor_act_model_layers_47_mlp_gate_proj/max_abs":2.171875,"train/train/layer_model_layers_27/grad/max_abs":0.00102996826171875,"train/train/tensor_act_model_layers_28_input_layernorm/norm":5792.613159185022,"train/train/layer_model_layers_49/grad/norm":0.043438175516109546,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/std":2.303161055595106e-05,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/max_abs":0.0001506805419921875,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/mean":0.0002193450927734375,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/max_abs":0.00035858154296875,"train/train/tensor_act_model_layers_33_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp/mean":0.0003266334533691406,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/norm":0.018105644710800912,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/max_abs":0.08740234375,"train/train/tensor_grad_model_layers_75_mlp_gate_proj_weight/mean":6.629852578043938e-08,"train/train/tensor_act_model_layers_63_self_attn_o_proj/std":0.08180223172145651,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/norm":8.125,"train/train/tensor_act_model_layers_15/max_abs":8.4375,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/norm":7.375,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/mean":5.9476587921381e-07,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/mean":-8.440110832452774e-08,"train/train/tensor_act_model_layers_8_self_attn_v_proj/std":0.24560609880465165,"train/train/tensor_act_model_layers_33/mean":-0.017333984375,"train/train/tensor_act_model_layers_9_mlp/max_abs":0.5,"train/train/layer_model_layers_1/act/std":0.6135478378557438,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/mean":1.194886863231659e-06,"train/train/tensor_act_model_layers_75_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp_up_proj/norm":3258.08500782689,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/std":0.04150390625,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/max_abs":0.19921875,"train/train/tensor_grad_model_layers_74_mlp_gate_proj_weight/max_abs":0.00064849853515625,"train/train/tensor_act_model_layers_52_post_attention_layernorm/mean":-0.00635528564453125,"train/train/tensor_act_model_layers_75_input_layernorm/max_abs":5.8125,"train/train/tensor_act_model_layers_86_mlp_gate_proj/mean":0.004009246826171875,"train/train/layer_model_layers_79/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/std":3.6023081296793554e-05,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/max_abs":0.18359375,"train/train/tensor_grad_model_layers_1_mlp_gate_proj_weight/norm":0.0347008347869851,"train/train/tensor_act_model_layers_33_mlp_down_proj/std":0.045532413565696546,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/norm":3.21875,"train/train/layer__model_layers_20/param/max_abs":1,"train/train/tensor_act_model_layers_77_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_o_proj/mean":0.001739501953125,"train/train/tensor_act_model_layers_27_input_layernorm/max_abs":6.46875,"train/train/tensor_act_model_layers_69_self_attn_k_proj/max_abs":5.21875,"train/train/tensor_act_model_layers_68_post_attention_layernorm/norm":5792.614257812963,"train/train/tensor_param_model_layers_77_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_11_mlp/norm":237.52016336661853,"train/train/tensor_act_model_layers_32_mlp_up_proj/max_abs":1.90625,"train/train/tensor_act_model_layers_76_self_attn_q_proj/max_abs":6.90625,"train/train/tensor_act_model_layers_10_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_52/act/norm":13521.87851176038,"train/train/tensor_param_model_layers_28_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_92_input_layernorm/max_abs":5.1875,"train/train/tensor_act_model_layers_45_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/norm":6.125,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/std":9.604965000036771e-05,"train/train/tensor_act_model_layers_7/mean":-0.03253173828125,"train/train/tensor_act_model_layers_26/max_abs":8.8125,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/mean":0.0002193450927734375,"train/train/tensor_act_model_layers_14_mlp_up_proj/norm":1901.8749798198817,"train/train/layer_model_layers_56/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp_up_proj/mean":-0.00394439697265625,"train/train/layer_model_layers_21/grad/mean":9.29978622382219e-08,"train/train/tensor_act_model_layers_51_mlp_gate_proj/mean":-2.0384788513183594e-05,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_74_self_attn_v_proj/mean":0.00557708740234375,"train/train/layer__model_layers_69/param/norm":22.97053564487559,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/norm":0.020337624783593248,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/max_abs":0.2578125,"train/train/tensor_act_model_layers_53_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer__model_layers_76/param/std":0.05908117071034139,"train/train/tensor_act_model_layers_30_self_attn_k_proj/norm":4348.271287757903,"train/train/layer_model_layers_85/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_q_proj/norm":6777.273889283741,"train/train/tensor_act_model_layers_91_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_61/param/max_abs":1,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn/std":0.1447910718658416,"train/train/tensor_grad_model_layers_73_mlp_gate_proj_weight/max_abs":0.00048828125,"train/train/tensor_param_model_layers_55_mlp_gate_proj_weight/mean":0.00013256072998046875,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/std":8.864891316281818e-05,"train/train/layer_model_layers_72/act/mean":0.010902455874851771,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/max_abs":0.142578125,"train/train/tensor_grad_model_layers_2_mlp_gate_proj_weight/mean":4.025641828775406e-07,"train/train/tensor_act_model_layers_0_mlp_gate_proj/mean":-0.021209716796875,"train/train/tensor_act_model_layers_44_post_attention_layernorm/std":1.0000002919695847,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/mean":3.4332275390625e-05,"train/train/tensor_act_model_layers_41/norm":7179.250746522958,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/std":0.03564453125,"train/train/tensor_grad_model_layers_19_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_mlp_up_proj/max_abs":2.171875,"train/train/tensor_act_model_layers_18_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/max_abs":0.10546875,"train/train/tensor_act_model_layers_77_self_attn_o_proj/norm":1636.79448506998,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/mean":-1.5995465219020844e-07,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn_v_proj/mean":0.0005655288696289062,"train/train/tensor_act_model_layers_41_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_2/grad/std":9.27505344558167e-05,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_34/act/norm":13535.912593960984,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/max_abs":0.1083984375,"train/train/tensor_act_model_layers_82_mlp_up_proj/max_abs":3.234375,"train/train/tensor_act_model_layers_89/max_abs":12.8125,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/norm":0.021981724880968726,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_52_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_7_self_attn_k_proj/std":0.8427753045115561,"train/train/tensor_param_model_layers_22_mlp_gate_proj_weight/max_abs":0.11572265625,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/norm":6.15625,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/norm":2.84375,"train/train/tensor_act_model_layers_16_self_attn_q_proj/norm":6191.207349385889,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_14_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/norm":5372.201378563099,"train/train/tensor_param_model_layers_3_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/norm":6.65625,"train/train/tensor_act_model_layers_80_self_attn_q_proj/max_abs":6.28125,"train/train/layer_model_layers_35/act/max_abs":9.4375,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/max_abs":0.232421875,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_49/grad/mean":1.98485399487424e-08,"train/train/tensor_param_model_layers_92_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_58/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/norm":0.0018544098379785177,"train/train/tensor_act_model_layers_39_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/std":0.03369140625,"train/train/tensor_act_model_layers_44_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp_down_proj/max_abs":1.7421875,"train/train/tensor_act_model_layers_43_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_43/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_mlp_down_proj/norm":426.6580840913031,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/norm":0.027094591380675964,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/max_abs":0.341796875,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/max_abs":0.00115203857421875,"train/train/tensor_act_model_layers_92_mlp_gate_proj/norm":6181.996203846414,"train/train/tensor_act_model_layers_0_post_attention_layernorm/mean":-0.0097503662109375,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/norm":3.78125,"train/train/tensor_param_model_layers_30_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_mlp_gate_proj_weight/mean":0.000179290771484375,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_28_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/norm":7.40625,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp/max_abs":0.34765625,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_o_proj/std":0.08838657418931434,"train/train/tensor_act_model_layers_81_self_attn_k_proj/std":1.0722714616793845,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/max_abs":0.224609375,"train/train/tensor_act_model_layers_57_mlp_up_proj/mean":-0.00909423828125,"train/train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn_o_proj/max_abs":1.0703125,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/mean":-4.744529724121094e-05,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/std":0.045654296875,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/max_abs":0.000347137451171875,"train/train/tensor_act_model_layers_74_self_attn/mean":-5.924701690673828e-05,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/mean":8.186907507479191e-08,"train/train/layer_model_layers_82/act/std":0.7941300408378092,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp/std":0.06677458564038419,"train/train/tensor_grad_model_layers_61_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp_up_proj/max_abs":2.140625,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/std":0.0245361328125,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_mlp_gate_proj_weight/max_abs":0.000579833984375,"train/train/tensor_act_model_layers_0_self_attn_k_proj/std":0.43798910069681357,"train/train/tensor_param_model_layers_73_mlp_gate_proj_weight/max_abs":0.16796875,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/max_abs":0.2109375,"train/train/tensor_grad_model_layers_38_mlp_gate_proj_weight/std":4.871508779431417e-05,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_74_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/std":0.04345703125,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/max_abs":0.00019931793212890625,"train/train/tensor_act_model_layers_55_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/std":1.023437549139706,"train/train/layer__model_layers_33/param/std":0.05077906459691126,"train/train/tensor_act_model_layers_64_post_attention_layernorm/mean":0.0021033287048339844,"train/train/tensor_act_model_layers_70_self_attn/std":0.1818866492320301,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/max_abs":0.0004558563232421875,"train/train/tensor_act_model_layers_33_self_attn_v_proj/std":0.3183594600531656,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/norm":0.04447817400048705,"train/train/tensor_act_model_layers_10_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/std":1.8395233097784614e-05,"train/train/tensor_act_model_layers_41_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_k_proj/mean":-0.01039886474609375,"train/train/layer_model_layers_18/grad/norm":0.03411680236374813,"train/train/tensor_act_model_layers_30_mlp_gate_proj/std":0.27197398770035025,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/max_abs":0.248046875,"train/train/tensor_act_model_layers_87_self_attn_o_proj/std":0.24267649068815048,"train/train/tensor_act_model_layers_77_self_attn_q_proj/max_abs":6.46875,"train/train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_q_proj/norm":5306.541469589778,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_20/act/mean":-0.0060846613986151555,"train/train/tensor_act_model_layers_40_mlp_up_proj/mean":-0.008941650390625,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_v_proj/mean":-0.0008411407470703125,"train/train/tensor_act_model_layers_42_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/max_abs":0.0003910064697265625,"train/train/tensor_act_model_layers_81_mlp/max_abs":1.53125,"train/train/layer__model_layers_4/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_gate_proj_weight/std":0.0003503280235742305,"train/train/tensor_act_model_layers_21_mlp/mean":0.0017547607421875,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/std":2.2937541253560764e-05,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/mean":-1.0326039046049118e-07,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/mean":1.4349818229675293e-05,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/std":0.047607421875,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/max_abs":0.15234375,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/norm":7.1875,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/norm":6.09375,"train/train/layer_model_layers_85/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp_down_proj/max_abs":0.37890625,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/mean":3.0710361897945404e-07,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/mean":-7.89295881986618e-08,"train/train/tensor_act_model_layers_30_post_attention_layernorm/max_abs":6.5625,"train/train/tensor_param_model_layers_69_mlp_gate_proj_weight/mean":0.000156402587890625,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/mean":-2.0582228899002075e-07,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp/norm":1446.4339918460219,"train/train/tensor_act_model_layers_22_self_attn_k_proj/std":0.7880878755718164,"train/train/tensor_param_model_layers_8_mlp_gate_proj_weight/mean":-2.7298927307128906e-05,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/mean":-6.24277163296938e-08,"train/train/tensor_act_model_layers_86_mlp_down_proj/norm":1565.8012737696185,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_89/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/mean":-0.00019931793212890625,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_o_proj/norm":416.3950798033205,"train/train/tensor_act_model_layers_9_self_attn_v_proj/norm":1632.7874078990733,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_down_proj/norm":407.2217637710753,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/norm":0.02447776114033758,"train/train/tensor_param_model_layers_49_mlp_gate_proj_weight/std":0.0322265625,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/max_abs":0.0001773834228515625,"train/train/tensor_act_model_layers_76_mlp_gate_proj/std":0.4785157444525589,"train/train/layer_model_layers_36/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/max_abs":0.00023746490478515625,"train/train/tensor_act_model_layers_84_mlp/max_abs":1.703125,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/mean":-0.0002841949462890625,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/norm":5.53125,"train/train/tensor_act_model_layers_85_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn/std":0.16504068831668958,"train/train/tensor_act_model_layers_62_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn_k_proj/max_abs":5.28125,"train/train/tensor_act_model_layers_5_self_attn_o_proj/mean":0.0011548995971679688,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/max_abs":0.0001354217529296875,"train/train/tensor_act_model_layers_62_input_layernorm/max_abs":5.8125,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_10/act/mean":-0.004570484161376953,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/std":0.00012447183349254136,"train/train/layer_model_layers_82/act/norm":17212.66232555377,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/std":0.03369140625,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/mean":-0.000102996826171875,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/mean":0.000293731689453125,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn/std":0.07007038840059213,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/norm":5.875,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/norm":5.65625,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/max_abs":0.00064849853515625,"train/train/tensor_act_model_layers_18_mlp_gate_proj/mean":-0.00182342529296875,"train/train/tensor_act_model_layers_40_self_attn_v_proj/std":0.4023438782647988,"train/train/tensor_act_model_layers_47_mlp_down_proj/std":0.06542974325900593,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_mlp_up_proj/norm":3811.626973032847,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/std":6.352678552281754e-05,"train/train/layer_model_layers_37/act/std":0.6527621656473421,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/norm":0.01439398056695341,"train/train/tensor_act_model_layers_49_mlp_down_proj/max_abs":0.65625,"train/train/tensor_act_model_layers_20_mlp_up_proj/std":0.2441406826376847,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/mean":-5.673617124557495e-06,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/max_abs":0.224609375,"train/train/tensor_act_model_layers_17/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn_v_proj/max_abs":1.7578125,"train/train/tensor_act_model_layers_22/std":1.2636782968083782,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_k_proj/norm":5272.1092325177115,"train/train/tensor_act_model_layers_35_self_attn_o_proj/norm":228.78234637280647,"train/train/tensor_act_model_layers_57_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_k_proj/std":0.912112202882203,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/mean":-0.0004596710205078125,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/norm":0.03226728467390496,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/std":6.306876967882806e-05,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75/std":1.582039373851633,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/norm":0.003813483998954448,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/std":0.036376953125,"train/train/tensor_act_model_layers_26_mlp_down_proj/max_abs":0.36328125,"train/train/tensor_act_model_layers_54_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_v_proj/max_abs":2.015625,"train/train/tensor_act_model_layers_80/mean":0.004225730895996094,"train/train/tensor_act_model_layers_15_input_layernorm/norm":5792.610839850212,"train/train/tensor_act_model_layers_9_mlp_gate_proj/std":0.2314453414220832,"train/train/tensor_act_model_layers_57/norm":7544.676983738497,"train/train/tensor_act_model_layers_88_mlp_down_proj/std":0.3247083954316358,"train/train/tensor_act_model_layers_26_self_attn_q_proj/mean":0.00331878662109375,"train/train/tensor_act_model_layers_29_mlp_up_proj/mean":-0.00180816650390625,"train/train/tensor_act_model_layers_91_self_attn/std":0.25244304289370284,"train/train/tensor_act_model_layers_20_input_layernorm/std":1.0000013052477366,"train/train/tensor_act_model_layers_38_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_14/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/max_abs":0.2158203125,"train/train/layer_model_layers_8/act/mean":-0.0013524549348013742,"train/train/tensor_act_model_layers_37_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_12/grad/max_abs":0.0010833740234375,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_83/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_v_proj/norm":1932.8493632693894,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/std":4.281766702079299e-05,"train/train/tensor_act_model_layers_62_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn/norm":595.4883635802908,"train/train/tensor_act_model_layers_82_self_attn_q_proj/norm":5891.667315066524,"train/train/tensor_act_model_layers_22_input_layernorm/norm":5792.613647466971,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/max_abs":0.0006561279296875,"train/train/tensor_act_model_layers_70_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/norm":7.21875,"train/train/layer__model_layers_13/param/norm":20.056670200490657,"train/train/layer__model_layers_45/param/norm":21.12690103317628,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_77_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/mean":5.718320608139038e-07,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_o_proj/std":0.10278873537730607,"train/train/layer_model_layers_35/grad/norm":0.034050034322756,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/norm":5.625,"train/train/tensor_act_model_layers_60_mlp_gate_proj/std":0.40088004715823355,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn/std":0.0639649144571616,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/max_abs":0.00058746337890625,"train/train/tensor_act_model_layers_91_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/std":6.82503293421588e-05,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/max_abs":0.1494140625,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_o_proj/max_abs":0.953125,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/max_abs":0.00049591064453125,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/std":0.028564453125,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/max_abs":0.1376953125,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/std":6.427045045315452e-05,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/std":0.04638671875,"train/train/tensor_grad_model_layers_33_mlp_gate_proj_weight/std":4.478137227372532e-05,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/max_abs":0.125,"train/train/tensor_act_model_layers_46_mlp/max_abs":0.474609375,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_q_proj/mean":0.002685546875,"train/train/layer_model_layers_50/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/norm":4.75,"train/train/tensor_act_model_layers_72_post_attention_layernorm/std":1.000001565515643,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/max_abs":0.000843048095703125,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/mean":1.6312114894390106e-06,"train/train/tensor_act_model_layers_73_post_attention_layernorm/std":1.0000023088842402,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/std":7.809823368300076e-05,"train/train/tensor_act_model_layers_63_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_mlp_gate_proj_weight/mean":-0.00010633468627929688,"train/train/tensor_act_model_layers_25_mlp/mean":0.00121307373046875,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/mean":-8.678436279296875e-05,"train/train/layer_model_layers_28/act/norm":14006.960867859154,"train/train/tensor_act_model_layers_48_self_attn/norm":467.0067649053334,"train/train/tensor_act_model_layers_41/std":1.2402394181997536,"train/train/tensor_act_model_layers_79_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/mean":-0.00021266937255859375,"train/train/tensor_act_model_layers_48_self_attn_v_proj/std":0.3730468888625656,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/norm":0.010550302372459196,"train/train/tensor_grad_model_layers_51_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_input_layernorm/max_abs":6.6875,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/mean":-7.706694304943085e-08,"train/train/tensor_act_model_layers_15_mlp_up_proj/max_abs":1.765625,"train/train/tensor_act_model_layers_68_self_attn_v_proj/mean":0.00362396240234375,"train/train/tensor_act_model_layers_7_mlp_gate_proj/std":0.21972666114566497,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/mean":-3.182794898748398e-07,"train/train/tensor_grad_model_layers_26_mlp_gate_proj_weight/std":4.132838314940366e-05,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/std":3.426226102781005e-05,"train/train/tensor_act_model_layers_1_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_input_layernorm/std":1.0000004259635717,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/max_abs":0.2236328125,"train/train/tensor_act_model_layers_10_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/norm":3.984375,"train/train/layer_model_layers_76/grad/max_abs":0.00102996826171875,"train/train/tensor_param_model_layers_22_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_gate_proj_weight/max_abs":0.1923828125,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/norm":6.78125,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/max_abs":0.000652313232421875,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82/max_abs":11.375,"train/train/tensor_act_model_layers_11_self_attn_k_proj/norm":5118.369377995717,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/std":0.05810546875,"train/train/layer_model_layers_13/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/norm":0.001298282491203402,"train/train/tensor_act_model_layers_19_self_attn_v_proj/max_abs":2.46875,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_50_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/norm":7.125,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/norm":5.78125,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/std":5.160589557042739e-05,"train/train/layer__model_layers_3/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn/norm":245.47060492655476,"train/train/tensor_act_model_layers_47_self_attn_k_proj/max_abs":5.15625,"train/train/tensor_param_model_layers_31_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_61_self_attn_v_proj/norm":2589.1957491582993,"train/train/layer_model_layers_73/grad/mean":7.941668007563503e-08,"train/train/tensor_act_model_layers_27_mlp_up_proj/std":0.26123185032073676,"train/train/tensor_act_model_layers_59_mlp_up_proj/mean":0.0172119140625,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/std":0.04443359375,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/max_abs":0.306640625,"train/train/tensor_param_model_layers_11_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp_up_proj/max_abs":2.65625,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/norm":0.015015795904213067,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/std":6.609107431631574e-05,"train/train/tensor_act_model_layers_80_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/std":0.03369140625,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/mean":-2.0710285753011703e-07,"train/train/tensor_act_model_layers_42_self_attn/mean":-0.0007524490356445312,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_mlp_gate_proj_weight/std":0.02490234375,"train/train/tensor_act_model_layers_80_mlp_up_proj/mean":-0.003452301025390625,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/max_abs":0.1748046875,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_gate_proj_weight/mean":0.000308990478515625,"train/train/layer__model_layers_79/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_gate_proj/max_abs":1.7578125,"train/train/tensor_act_model_layers_36_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/std":0.06005859375,"train/train/tensor_act_model_layers_39/norm":7184.4315844342555,"train/train/tensor_act_model_layers_89_mlp/std":0.3701185554063771,"train/train/tensor_act_model_layers_59_mlp_down_proj/max_abs":0.6796875,"train/train/tensor_act_model_layers_21_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_gate_proj/std":0.6250003715510307,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/norm":0.014886041676387107,"train/train/tensor_act_model_layers_16_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_v_proj/max_abs":1.65625,"train/train/tensor_act_model_layers_93_mlp_up_proj/mean":0.00315093994140625,"train/train/tensor_act_model_layers_53_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_32/grad/mean":7.616794797448025e-08,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_1/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/mean":0.0003719329833984375,"train/train/tensor_act_model_layers_24_mlp_up_proj/max_abs":1.9296875,"train/train/tensor_param_model_layers_23_mlp_gate_proj_weight/norm":4.78125,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/mean":7.104873657226562e-05,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/mean":3.273598849773407e-07,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/std":0.00011502714312832544,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/max_abs":0.1806640625,"train/train/tensor_act_model_layers_83_mlp_down_proj/std":0.21997335905489115,"train/train/tensor_param_model_layers_83_mlp_gate_proj_weight/std":0.04345703125,"train/train/tensor_act_model_layers_2_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp_gate_proj/max_abs":2.265625,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/mean":-0.0001506805419921875,"train/train/layer__model_layers_86/param/mean":0.0017143642288660295,"train/train/tensor_act_model_layers_72_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/norm":0.004562939255289291,"train/train/tensor_act_model_layers_1/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_down_proj/norm":219.4234072503138,"train/train/tensor_act_model_layers_14_mlp/max_abs":0.52734375,"train/train/tensor_act_model_layers_11_input_layernorm/mean":-0.029998779296875,"train/train/tensor_act_model_layers_85_input_layernorm/max_abs":5.84375,"train/train/tensor_act_model_layers_39_self_attn_v_proj/max_abs":2.0625,"train/train/tensor_act_model_layers_38_self_attn_o_proj/max_abs":1.4765625,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/std":7.702621102885005e-05,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_o_proj/max_abs":3.1875,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/norm":0.022825679629800432,"train/train/tensor_act_model_layers_48_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_post_attention_layernorm/norm":5792.617309575389,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_o_proj/mean":-0.0007534027099609375,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/norm":6.9375,"train/train/tensor_grad_model_layers_46_mlp_gate_proj_weight/mean":1.500011421740055e-07,"train/train/tensor_act_model_layers_44_self_attn_v_proj/mean":0.002704620361328125,"train/train/tensor_act_model_layers_49_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/norm":0.01364801231730812,"train/train/layer_model_layers_15/act/mean":-0.0036935976573399137,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/mean":-1.8399441614747047e-07,"train/train/layer_model_layers_1/grad/max_abs":0.004974365234375,"train/train/layer_model_layers_66/act/norm":15017.08254403713,"train/train/tensor_act_model_layers_93_self_attn/max_abs":2.828125,"train/train/tensor_act_model_layers_77_self_attn_v_proj/norm":2698.4868754240542,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp_down_proj/std":0.04199220052768469,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/mean":0.0002231597900390625,"train/train/tensor_act_model_layers_10_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp_gate_proj/mean":0.0014057159423828125,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp_down_proj/mean":0.00061798095703125,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_post_attention_layernorm/std":1.00000129717871,"train/train/tensor_act_model_layers_25_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_mlp_gate_proj_weight/norm":0.01667388421976753,"train/train/tensor_grad_model_layers_42_mlp_gate_proj_weight/max_abs":0.000545501708984375,"train/train/tensor_act_model_layers_20_mlp/mean":0.001190185546875,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/std":1.76533560191675e-05,"train/train/tensor_act_model_layers_69_post_attention_layernorm/norm":5792.604736331927,"train/train/tensor_act_model_layers_29_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/mean":1.1175870895385742e-06,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/std":3.902967356064291e-05,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/max_abs":0.0015106201171875,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/mean":3.0873343348503113e-07,"train/train/tensor_act_model_layers_1_self_attn_v_proj/norm":1216.7035005201042,"train/train/tensor_act_model_layers_63_self_attn_q_proj/std":0.8378929191504596,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/max_abs":0.1962890625,"train/train/tensor_act_model_layers_10_self_attn_k_proj/std":0.9462905874919119,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/mean":3.841705620288849e-09,"train/train/tensor_act_model_layers_78_mlp_down_proj/max_abs":1.296875,"train/train/tensor_grad_model_layers_39_mlp_gate_proj_weight/max_abs":0.000499725341796875,"train/train/tensor_act_model_layers_35_post_attention_layernorm/mean":-0.0106964111328125,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/std":5.3320948248607075e-05,"train/train/layer__model_layers_17/param/norm":20.36315798507196,"train/train/layer_model_layers_25/act/max_abs":8.75,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/std":3.124261477250392e-05,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/mean":-1.230509951710701e-07,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_input_layernorm/mean":-0.0138092041015625,"train/train/tensor_param_model_layers_9_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_82_self_attn_v_proj/mean":-0.0090484619140625,"train/train/layer_model_layers_38/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn_q_proj/mean":-0.124755859375,"train/train/tensor_act_model_layers_93_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_post_attention_layernorm/max_abs":6.5,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_56/act/std":0.6619307616087398,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_v_proj/max_abs":3.03125,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/mean":8.162169251590967e-08,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/std":7.05103254604983e-05,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/mean":0.0002002716064453125,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/std":0.044921875,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/std":2.0712296711660098e-05,"train/train/tensor_act_model_layers_76_mlp_up_proj/norm":3895.4946292268637,"train/train/tensor_param_model_layers_77_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_37/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_gate_proj_weight/mean":-6.9141387939453125e-06,"train/train/tensor_act_model_layers_92_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/mean":-2.0742416381835938e-05,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/mean":-1.725275069475174e-07,"train/train/layer_model_layers_90/grad/max_abs":0.001190185546875,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/std":4.741384340728068e-05,"train/train/tensor_act_model_layers_31_self_attn_k_proj/std":0.7519558217570015,"train/train/tensor_act_model_layers_19_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_42/param/mean":0.0015911869995307625,"train/train/tensor_act_model_layers_10/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_36/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18/mean":-0.031524658203125,"train/train/tensor_param_model_layers_7_mlp_gate_proj_weight/max_abs":0.1044921875,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/max_abs":0.1865234375,"train/train/tensor_grad_model_layers_83_mlp_gate_proj_weight/max_abs":0.000759124755859375,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/norm":0.005871977763015218,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/mean":6.459653377532959e-06,"train/train/layer_model_layers_76/grad/norm":0.06321586572566704,"train/train/tensor_grad_model_layers_10_mlp_gate_proj_weight/norm":0.015032532748252182,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/norm":5.90625,"train/train/tensor_param_model_layers_37_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_68/act/max_abs":10.875,"train/train/tensor_act_model_layers_13_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/mean":0.00046539306640625,"train/train/tensor_param_model_layers_68_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_mlp_up_proj/mean":0.0115203857421875,"train/train/tensor_act_model_layers_39_post_attention_layernorm/std":1.0000004259635717,"train/train/tensor_param_model_layers_46_mlp_gate_proj_weight/norm":5.71875,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_86/act/norm":18645.565718419508,"train/train/tensor_act_model_layers_66_input_layernorm/max_abs":5.5625,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/mean":-1.5195109881460667e-07,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_mlp_gate_proj_weight/std":4.841536779258884e-05,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp/max_abs":1.4453125,"train/train/tensor_act_model_layers_77_mlp/max_abs":1.2734375,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/norm":6.59375,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/max_abs":0.00075531005859375,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/std":9.183988939484457e-05,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/max_abs":0.00012874603271484375,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_32/param/max_abs":1,"train/train/layer_model_layers_80/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/std":0.057373046875,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/norm":5.1875,"train/train/tensor_act_model_layers_75/mean":0.004748344421386719,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/max_abs":0.1416015625,"train/train/tensor_act_model_layers_30/std":1.2421888795281195,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/mean":-0.000179290771484375,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_gate_proj_weight/mean":-8.42846930027008e-08,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_2/act/std":0.6841186824059515,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/mean":4.4563785195350647e-07,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/std":0.031494140625,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_post_attention_layernorm/norm":5792.610229492243,"train/train/tensor_param_model_layers_58_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn/std":0.23734831686790042,"train/train/tensor_act_model_layers_50_self_attn/std":0.0933866118792555,"train/train/tensor_act_model_layers_58_self_attn_k_proj/std":0.8398442734117872,"train/train/tensor_act_model_layers_67_input_layernorm/std":1.0000009375721828,"train/train/tensor_act_model_layers_41_self_attn_q_proj/std":0.9169980598399803,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/std":6.843998783644483e-05,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_k_proj/norm":4779.031761700481,"train/train/tensor_act_model_layers_41_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_gate_proj_weight/std":8.035762893507475e-05,"train/train/layer_model_layers_50/grad/mean":1.4787542745587234e-07,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/mean":1.8894672393798828e-05,"train/train/tensor_act_model_layers_59_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/mean":0.0003509521484375,"train/train/layer_model_layers_30/grad/mean":8.24817311376124e-08,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/std":0.04443359375,"train/train/layer__model_layers_91/param/std":0.06575259895642467,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/max_abs":0.11083984375,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_44_self_attn_q_proj/norm":5277.679640504238,"train/train/tensor_act_model_layers_59/max_abs":9.9375,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_input_layernorm_weight/mean":1,"train/train/layer_model_layers_43/act/max_abs":9.3125,"train/train/tensor_act_model_layers_28_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/norm":7.34375,"train/train/tensor_act_model_layers_83_input_layernorm/std":1.000001904225434,"train/train/tensor_act_model_layers_57_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_22_mlp_up_proj/max_abs":1.6640625,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/norm":0.0022851405795285495,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/mean":-4.1961669921875e-05,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/norm":4.75,"train/train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_self_attn_k_proj/std":0.8085940625643934,"train/train/tensor_act_model_layers_89_self_attn_o_proj/max_abs":1.984375,"train/train/layer_model_layers_21/grad/max_abs":0.0015106201171875,"train/train/tensor_param_model_layers_45_mlp_gate_proj_weight/norm":5.65625,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/std":0.0400390625,"train/train/tensor_act_model_layers_4_self_attn/max_abs":1.078125,"train/train/tensor_act_model_layers_61_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp/mean":-0.00015848875045776367,"train/train/layer__model_layers_19/param/norm":20.21044675911755,"train/train/tensor_act_model_layers_26/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/max_abs":0.000148773193359375,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/norm":0.007695484014235995,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/std":4.407124198777962e-05,"train/train/tensor_act_model_layers_12_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_q_proj/norm":5725.145594638263,"train/train/tensor_act_model_layers_62_mlp/norm":585.4367700284329,"train/train/layer__model_layers_60/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/max_abs":0.0018157958984375,"train/train/tensor_act_model_layers_88_mlp_gate_proj/max_abs":3.90625,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_down_proj/mean":0.000797271728515625,"train/train/tensor_grad_model_layers_5_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/mean":-0.0003261566162109375,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/norm":0.028874983502668154,"train/train/tensor_grad_model_layers_74_mlp_gate_proj_weight/norm":0.02695481391812081,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/mean":4.141838871873915e-09,"train/train/tensor_act_model_layers_80_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/norm":0.003084748286459014,"train/train/tensor_act_model_layers_3_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_mlp_gate_proj_weight/norm":0.014732469492135794,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp_down_proj/norm":253.3229871528563,"train/train/layer__model_layers_58/param/std":0.05508692695388735,"train/train/tensor_act_model_layers_41_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer__model_layers_93/param/std":0.06645831586241495,"train/train/layer__model_layers_5/param/mean":0.0015324251886091069,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/mean":-1.8579885363578796e-07,"train/train/tensor_act_model_layers_16_mlp/std":0.03564495862243693,"train/train/tensor_act_model_layers_36_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_mlp_gate_proj_weight/max_abs":0.1279296875,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/std":0.052001953125,"train/train/tensor_act_model_layers_37_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/std":0.0341796875,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_mlp_gate_proj_weight/mean":2.5153160095214844e-05,"train/train/tensor_act_model_layers_62/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/max_abs":0.1123046875,"train/train/tensor_act_model_layers_80_self_attn_k_proj/norm":5626.797790110625,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/max_abs":0.138671875,"train/train/tensor_param_model_layers_34_mlp_gate_proj_weight/std":0.029052734375,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/mean":-0.000606536865234375,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/mean":-0.031463623046875,"train/train/tensor_act_model_layers_70_self_attn/max_abs":2.265625,"train/train/tensor_param_model_layers_21_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/max_abs":0.16015625,"train/train/tensor_act_model_layers_33_mlp_gate_proj/std":0.28320315378750044,"train/train/layer__model_layers_24/param/max_abs":1,"train/train/tensor_act_model_layers_39_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_q_proj/std":1.2226637087463437,"train/train/tensor_act_model_layers_87/mean":0.005519866943359375,"train/train/tensor_act_model_layers_82_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/mean":6.723403930664062e-05,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/mean":3.17990779876709e-05,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/max_abs":0.000530242919921875,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_mlp_gate_proj_weight/mean":5.364418029785156e-05,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/max_abs":0.00028228759765625,"train/train/layer_model_layers_38/act/max_abs":9.4375,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/std":3.6770196787649566e-05,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/std":0.03662109375,"train/train/tensor_act_model_layers_31_input_layernorm/std":1.0000009093196653,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/std":9.042235377200226e-05,"train/train/layer_model_layers_31/grad/std":4.553643576265129e-05,"train/train/tensor_act_model_layers_23_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/mean":6.341934204101562e-05,"train/train/tensor_act_model_layers_39_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/max_abs":0.00038909912109375,"train/train/tensor_param_model_layers_44_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/std":0.0272216796875,"train/train/layer_model_layers_67/grad/std":7.167092317597898e-05,"train/train/tensor_act_model_layers_66_input_layernorm/norm":5792.6121826182925,"train/train/tensor_act_model_layers_80_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp/norm":532.4672282305404,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/norm":0.0006197004767267881,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/norm":0.0010392162376874439,"train/train/layer__model_layers_93/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/max_abs":0.000400543212890625,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/norm":6.15625,"train/train/layer_model_layers_67/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp/max_abs":0.69921875,"train/train/tensor_act_model_layers_25/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/max_abs":0.140625,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/mean":-0.000644683837890625,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/max_abs":0.1455078125,"train/train/tensor_act_model_layers_18_self_attn/std":0.039429669432663426,"train/train/layer_model_layers_71/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp_up_proj/mean":0.0012531280517578125,"train/train/tensor_grad_model_layers_65_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_25/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/norm":0.00901994505503407,"train/train/tensor_act_model_layers_32_self_attn/max_abs":1.078125,"train/train/layer__model_layers_34/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/mean":-5.494803190231323e-08,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/std":5.950745259437939e-05,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn/std":0.04169219450259611,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/norm":0.02347188652301374,"train/train/tensor_act_model_layers_37/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/norm":0.007027943814004274,"train/train/tensor_act_model_layers_76_self_attn/norm":1001.2058568754359,"train/train/tensor_act_model_layers_66_self_attn/max_abs":4,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/mean":9.5367431640625e-05,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_71_self_attn_k_proj/mean":0.031829833984375,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/norm":0.0032004014217437166,"train/train/layer_model_layers_60/act/std":0.647747467144436,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_42_mlp_gate_proj_weight/norm":5.59375,"train/train/tensor_act_model_layers_88_self_attn_v_proj/norm":2665.470797809035,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/norm":7.9375,"train/train/tensor_act_model_layers_17_self_attn_k_proj/std":0.8808642046810804,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/norm":0.0009979006708942424,"train/train/tensor_act_model_layers_92_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn/std":0.10131873739386307,"train/train/tensor_act_model_layers_22_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/std":4.490503970881343e-05,"train/train/tensor_act_model_layers_65_mlp/std":0.11718755282150012,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_v_proj/std":0.4843752414828706,"train/train/tensor_act_model_layers_45_mlp/std":0.06048593385180993,"train/train/tensor_act_model_layers_58_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_v_proj/std":0.47119246329601655,"train/train/tensor_act_model_layers_47_self_attn_v_proj/std":0.36572370064284526,"train/train/tensor_act_model_layers_48_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_mlp_gate_proj_weight/max_abs":0.1611328125,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/mean":2.713495632633567e-07,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/max_abs":0.2353515625,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/max_abs":0.00017261505126953125,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/std":5.1980474886759426e-05,"train/train/tensor_act_model_layers_45_mlp_down_proj/max_abs":0.47265625,"train/train/tensor_act_model_layers_17_mlp_up_proj/mean":0.0025482177734375,"train/train/tensor_act_model_layers_84_self_attn_v_proj/max_abs":2.625,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/mean":5.4836273193359375e-05,"train/train/tensor_act_model_layers_40_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56/mean":-0.003498077392578125,"train/train/tensor_act_model_layers_82_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model/max_abs":5.09375,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/mean":7.488415576517582e-08,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/norm":0.030060542528658883,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/norm":0.014439362339769966,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/norm":0.00807355780773029,"train/train/tensor_grad_model_layers_81_mlp_gate_proj_weight/mean":6.435438990592957e-07,"train/train/tensor_act_model_layers_15_self_attn_o_proj/norm":352.022946903137,"train/train/tensor_act_model_layers_59_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/std":0.20898444159390545,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/std":0.00014745588047778243,"train/train/tensor_act_model_layers_79_self_attn_q_proj/std":0.959962830142671,"train/train/tensor_act_model_layers_87_mlp_up_proj/mean":-0.013214111328125,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/std":4.488297993800516e-05,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_input_layernorm/mean":-0.029937744140625,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_input_layernorm/max_abs":5.65625,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/max_abs":0.00038909912109375,"train/train/tensor_act_model_layers_74_self_attn/norm":1198.9102010576203,"train/train/tensor_act_model_layers_26_self_attn_o_proj/max_abs":0.96484375,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/mean":0.00013637542724609375,"train/train/tensor_act_model_layers_16_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/mean":5.078315734863281e-05,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/max_abs":0.1826171875,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/std":5.915446346195511e-05,"train/train/tensor_act_model_layers_51_self_attn_o_proj/norm":406.2135493347889,"train/train/tensor_act_model_layers_56_input_layernorm/std":1.0000003207146104,"train/train/tensor_act_model_layers_46_self_attn_v_proj/mean":-0.00457763671875,"train/train/layer_model_layers_2/act/mean":0.007457967315401349,"train/train/layer__model_layers_1/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/mean":-0.0002574920654296875,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/max_abs":0.283203125,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/mean":-1.9587576389312744e-05,"train/train/tensor_act_model_layers_20_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/std":0.040283203125,"train/train/tensor_act_model_layers_79_self_attn_v_proj/mean":0.003936767578125,"train/train/tensor_param_model_layers_91_mlp_gate_proj_weight/mean":-3.3855438232421875e-05,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/max_abs":0.0005035400390625,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/std":0.045166015625,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/max_abs":0.00067901611328125,"train/train/tensor_act_model_layers_38_mlp_gate_proj/norm":2539.7305299972263,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/norm":7.09375,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/norm":0.0008551539685387309,"train/train/layer_model_layers_28/grad/max_abs":0.0022735595703125,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/max_abs":0.0001583099365234375,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/std":0.052001953125,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_gate_proj_weight/mean":-5.916808731853962e-08,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_v_proj/max_abs":2.734375,"train/train/tensor_act_model_layers_77_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_mlp_up_proj/max_abs":1.921875,"train/train/tensor_act_model_layers_48_self_attn_o_proj/norm":467.0067649053334,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/std":2.086799850662094e-05,"train/train/tensor_act_model_layers_36/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_k_proj/std":0.9570353566295721,"train/train/tensor_act_model_layers_7_mlp/norm":246.29436084516988,"train/train/tensor_act_model_layers_47_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_mlp_gate_proj_weight/std":4.968021922946404e-05,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/max_abs":0.2431640625,"train/train/tensor_act_model_layers_79_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_23/act/mean":-0.003281661442347935,"train/train/tensor_act_model_layers_91_mlp/norm":2993.8624633984227,"train/train/tensor_act_model_layers_0_mlp/std":1.2714891243185245,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/mean":-1.739244908094406e-07,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/mean":-4.145083948969841e-06,"train/train/tensor_act_model_layers_12_input_layernorm/max_abs":5.71875,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20/max_abs":8.5,"train/train/layer_model_layers_86/act/std":0.8602145039233926,"train/train/tensor_act_model_layers_29_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_q_proj/mean":-0.0718994140625,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/mean":8.866190910339355e-06,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/mean":-0.00028228759765625,"train/train/layer_model_layers_45/act/std":0.6132213540097429,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/mean":2.1576881408691406e-05,"train/train/tensor_act_model_layers_41_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_self_attn_v_proj/std":0.46631154438902744,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/std":0.0264892578125,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/std":5.086442707538634e-05,"train/train/tensor_act_model_layers_18_mlp/max_abs":0.470703125,"train/train/tensor_act_model_layers_10/mean":-0.032257080078125,"train/train/tensor_act_model_layers_51_self_attn_o_proj/std":0.07020243010870277,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/mean":-0.000347137451171875,"train/train/tensor_act_model_layers_0_self_attn/std":0.03344728437380979,"train/train/tensor_act_model_layers_60_mlp/std":0.10009766871609262,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/std":0.044677734375,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/mean":4.348112270236015e-08,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/max_abs":0.000507354736328125,"train/train/tensor_act_model_layers_93/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/norm":0.0076818872873377,"train/train/tensor_act_model_layers_28_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_q_proj/max_abs":6.53125,"train/train/tensor_act_model_layers_0/max_abs":8.5625,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/std":0.00010216448845770895,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/max_abs":0.3125,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/norm":3.921875,"train/train/tensor_act_model_layers_74_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_73/grad/norm":0.06322863111308524,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/max_abs":0.000942230224609375,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_69/act/std":0.6869301612552475,"train/train/layer__model_layers_46/param/mean":0.0013848600819032762,"train/train/tensor_act_model_layers_72_mlp_gate_proj/std":0.46093751168099484,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/norm":4.21875,"train/train/tensor_param_model_layers_82_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_92_self_attn_k_proj/std":1.1054759210684204,"train/train/tensor_act_model_layers_10_mlp_gate_proj/norm":1898.7135236192728,"train/train/tensor_param_model_layers_85_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/std":2.003210676461373e-05,"train/train/tensor_act_model_layers_9_self_attn_v_proj/max_abs":2.265625,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/mean":7.992639439180493e-08,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/norm":7.15625,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/std":1.5457675638655996e-05,"train/train/tensor_act_model_layers_29_self_attn_k_proj/norm":5179.398262350371,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/std":2.406162451388641e-05,"train/train/layer_model_layers_19/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/norm":4.71875,"train/train/tensor_act_model_layers_36_self_attn/mean":-0.0008983612060546875,"train/train/tensor_act_model_layers_34_self_attn_k_proj/norm":4717.246205798168,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/std":0.00014308810213516036,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/mean":3.484543412923813e-06,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/max_abs":0.0002574920654296875,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/max_abs":9.250640869140625e-05,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/max_abs":0.203125,"train/train/tensor_act_model_layers_31_self_attn_q_proj/mean":0.04168701171875,"train/train/tensor_act_model_layers_24_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_57/act/mean":0.01048055716923305,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/norm":6.375,"train/train/layer_model_layers_15/grad/max_abs":0.001251220703125,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/std":0.0238037109375,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/mean":0.000240325927734375,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_mlp_gate_proj_weight/norm":6.125,"train/train/tensor_act_model_layers_56/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_o_proj/norm":221.1266481226035,"train/train/tensor_act_model_layers_86_self_attn/mean":5.435943603515625e-05,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/norm":0.0012938404940489635,"train/train/tensor_act_model_layers_54_self_attn_q_proj/std":0.9775417301168255,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_o_proj/std":0.1447910718658416,"train/train/tensor_act_model_layers_14/frac_near_user_limit":0,"train/train/layer_model_layers_59/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/norm":0.0008217529970700178,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/mean":-0.0009202957153320312,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_v_proj/std":0.2753908311448503,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/max_abs":0.000335693359375,"train/train/tensor_act_model_layers_83_mlp_up_proj/mean":0.00380706787109375,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_v_proj/norm":2046.7838214594933,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/mean":2.5214831111952662e-08,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/norm":0.019296571366241782,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/norm":5.0625,"train/train/layer_model_layers_5/grad/std":6.069142272863432e-05,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/norm":3.109375,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/norm":0.004191947896681029,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/norm":0.015052437717617438,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/max_abs":0.000335693359375,"train/train/tensor_act_model_layers_68_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_45/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_31/act/max_abs":9,"train/train/tensor_param_model_layers_56_mlp_gate_proj_weight/mean":-0.000133514404296875,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_57/grad/std":6.89278723629783e-05,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/norm":3.46875,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/max_abs":0.267578125,"train/train/tensor_act_model_layers_8_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/max_abs":0.2197265625,"train/train/tensor_grad_model_layers_83_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_mlp_gate_proj_weight/norm":0.026371849495624208,"train/train/tensor_act_model_layers_93_mlp_down_proj/mean":-0.019775390625,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/mean":-0.00016689300537109375,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_34_mlp_gate_proj_weight/mean":-2.561137080192566e-09,"train/train/layer_model_layers_85/act/mean":0.003734792981828962,"train/train/tensor_param_model_layers_79_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/std":0.0478515625,"train/train/tensor_act_model_layers_18_post_attention_layernorm/mean":-0.02716064453125,"train/train/tensor_act_model_layers_0_self_attn_q_proj/std":0.3662122408525797,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn/max_abs":1.0546875,"train/train/tensor_param_model_layers_34_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_mlp_gate_proj/norm":2879.1591328246136,"train/train/layer__model_layers_17/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_78/grad/norm":0.06520249170185162,"train/train/tensor_act_model_layers_39_mlp_up_proj/norm":2555.8110942381513,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/mean":0.00022411346435546875,"train/train/tensor_act_model_layers_2_self_attn_k_proj/std":1.1191458111121895,"train/train/tensor_param_model_layers_46_mlp_gate_proj_weight/std":0.031494140625,"train/train/tensor_act_model_layers_92_mlp_down_proj/norm":3724.876187052388,"train/train/tensor_act_model_layers_47_self_attn_q_proj/mean":-0.0128936767578125,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/std":9.372452935798967e-05,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/norm":0.004832846575067114,"train/train/tensor_act_model_layers_44_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/std":0.03515625,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/norm":0.004399826250655512,"train/train/tensor_param_model_layers_57_mlp_gate_proj_weight/std":0.034912109375,"train/train/tensor_act_model_layers_89_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/norm":4.78125,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/std":0.00010209787958079765,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp/norm":263.7055901079704,"train/train/tensor_act_model_layers_51_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp_down_proj/norm":239.61430995911246,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/std":0.02734375,"train/train/tensor_act_model_layers_69_post_attention_layernorm/mean":0.0075531005859375,"train/train/tensor_act_model_layers_38_mlp/std":0.05541992828143223,"train/train/tensor_act_model_layers_28_mlp/max_abs":0.357421875,"train/train/tensor_act_model_layers_93_input_layernorm/norm":5792.6130371102545,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_11_mlp_down_proj/max_abs":0.7265625,"train/train/layer__model_layers_79/param/max_abs":1,"train/train/layer__model_layers_78/param/norm":24.01893377499301,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/std":8.853725630363653e-05,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/mean":0.00031280517578125,"train/train/tensor_act_model_layers_79_self_attn_q_proj/norm":5557.431848144109,"train/train/tensor_act_model_layers_17/mean":-0.032196044921875,"train/train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_84_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_58_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_mlp_gate_proj_weight/norm":4.59375,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/std":0.0556640625,"train/train/tensor_act_model_layers_33_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_k_proj/norm":4029.918840115007,"train/train/tensor_act_model_layers_46_self_attn_q_proj/norm":5616.9436421063,"train/train/layer_model_layers_8/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/mean":-0.00020503997802734375,"train/train/tensor_act_model_layers_35_post_attention_layernorm/norm":5792.607666018737,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/mean":-0.0004138946533203125,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/norm":4.4375,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/max_abs":7.534027099609375e-05,"train/train/tensor_act_model_layers_47_self_attn_v_proj/norm":2121.3850931830207,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/std":2.5513964148233884e-05,"train/train/tensor_act_model_layers_17_mlp/std":0.043701519241331026,"train/train/tensor_act_model_layers_57/std":1.3027398196256013,"train/train/tensor_act_model_layers_25_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/max_abs":0.099609375,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/std":0.054931640625,"train/train/tensor_param_model_layers_77_mlp_gate_proj_weight/std":0.04150390625,"train/train/tensor_act_model_layers_25_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/max_abs":0.0009002685546875,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/std":0.048095703125,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_61_input_layernorm/std":1.0000005788965864,"train/train/tensor_act_model_layers_14_self_attn_v_proj/norm":1645.9543412104824,"train/train/layer_model_layers_49/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/mean":1.7159618437290192e-07,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/norm":0.006799704028711304,"train/train/tensor_act_model_layers_28_post_attention_layernorm/max_abs":6.5,"train/train/tensor_act_model_layers_88_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/mean":-5.340576171875e-05,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/std":5.70128007730341e-05,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_8/act/norm":13970.082956222559,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/norm":5.90625,"train/train/tensor_grad_model_layers_25_mlp_gate_proj_weight/std":3.9595752667658e-05,"train/train/tensor_act_model_layers_63_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/norm":0.02447943518200505,"train/train/tensor_act_model_layers_23_self_attn_v_proj/norm":1549.189227142156,"train/train/tensor_act_model_layers_1_input_layernorm/norm":5792.601196292123,"train/train/tensor_act_model_layers_46_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/mean":9.261071681976318e-06,"train/train/tensor_act_model_layers_73_self_attn_k_proj/mean":-0.013275146484375,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/max_abs":0.00072479248046875,"train/train/tensor_act_model_layers_69_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_post_attention_layernorm/std":1.000001583712784,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/mean":4.761386662721634e-08,"train/train/tensor_act_model_layers_82_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_gate_proj/norm":4722.68619028496,"train/train/tensor_act_model_layers_5/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_up_proj/max_abs":2.734375,"train/train/tensor_act_model_layers_34_input_layernorm/std":1.0000007211926587,"train/train/tensor_act_model_layers_23_self_attn_q_proj/mean":0.0097198486328125,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/max_abs":0.00023174285888671875,"train/train/layer_model_layers_48/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/std":0.03759765625,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/std":0.035400390625,"train/train/tensor_param_model_layers_79_mlp_gate_proj_weight/mean":-8.249282836914062e-05,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/max_abs":0.000225067138671875,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/max_abs":0.008056640625,"train/train/tensor_act_model_layers_14_self_attn_q_proj/mean":-0.00746917724609375,"train/train/tensor_act_model_layers_26_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_46/param/max_abs":1,"train/train/tensor_act_model_layers_2_post_attention_layernorm/mean":-0.022186279296875,"train/train/tensor_act_model_layers_45_self_attn_o_proj/std":0.06750612821338775,"train/train/tensor_act_model_layers_74_mlp_down_proj/mean":0.0016307830810546875,"train/train/layer__model_layers_37/param/max_abs":1,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/mean":1.041218638420105e-06,"train/train/tensor_grad_model_layers_66_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_13/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_gate_proj_weight/std":0.02392578125,"train/train/tensor_act_model_layers_27_self_attn_q_proj/std":0.9541030962354813,"train/train/tensor_act_model_layers_34/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_post_attention_layernorm/mean":-0.0170440673828125,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/norm":0.02615404518068435,"train/train/tensor_act_model_layers_74_input_layernorm/std":1.0000019533016735,"train/train/tensor_act_model_layers_49_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/mean":0.000141143798828125,"train/train/tensor_act_model_layers_91_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/std":8.857421091707395e-05,"train/train/tensor_act_model_layers_74_mlp/norm":864.5204021687772,"train/train/tensor_act_model_layers_76_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/std":0.02734375,"train/train/tensor_param_model_layers_87_mlp_gate_proj_weight/norm":8.6875,"train/train/tensor_param_model_layers_15_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_up_proj/std":0.2695313286630025,"train/train/tensor_act_model_layers_92_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_k_proj/mean":-0.15869140625,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp_down_proj/std":0.07055707892651966,"train/train/layer_model_layers_82/grad/mean":-1.1621627160427909e-07,"train/train/tensor_act_model_layers_40_self_attn_q_proj/max_abs":5.96875,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_51_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_q_proj/mean":-0.023773193359375,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/norm":0.002593219865918392,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/mean":4.614339559338987e-07,"train/train/layer_model_layers_47/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_q_proj/mean":0.0352783203125,"train/train/tensor_act_model_layers_81_self_attn_k_proj/mean":0.0830078125,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/norm":0.005847481737100167,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/norm":0.005905387932822047,"train/train/layer_model_layers_42/act/norm":13837.131467388263,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/mean":-5.564652383327484e-08,"train/train/tensor_act_model_layers_72_mlp_up_proj/max_abs":2.5625,"train/train/tensor_act_model_layers_37_self_attn_v_proj/norm":2296.5399811767084,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/norm":7.03125,"train/train/tensor_param_model_layers_50_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/mean":1.0181975085288286e-07,"train/train/tensor_act_model_layers_24_self_attn_k_proj/norm":4459.8801016367615,"train/train/tensor_act_model_layers_55_self_attn_v_proj/norm":1928.8652966096145,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/norm":0.0208587590673142,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/norm":5.59375,"train/train/layer_model_layers_4/grad/norm":0.0630066507790587,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/norm":0.0010087782817912612,"train/train/tensor_act_model_layers_4_post_attention_layernorm/std":1.000000093132253,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/norm":0.024609356834768222,"train/train/tensor_act_model_layers_7_self_attn_q_proj/std":0.9570312889254815,"train/train/tensor_grad_model_layers_37_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/max_abs":0.197265625,"train/train/tensor_act_model_layers_12_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp/mean":0.0002980232238769531,"train/train/layer_model_layers_45/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/norm":4.0625,"train/train/layer_model_layers_34/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_74/act/max_abs":11,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/std":3.5392602791754414e-05,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/mean":-0.00016021728515625,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/max_abs":0.1044921875,"train/train/tensor_act_model_layers_41/mean":-0.01708984375,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/std":0.0322265625,"train/train/tensor_grad_model_layers_18_mlp_gate_proj_weight/norm":0.014885728855698927,"train/train/tensor_act_model_layers_24_self_attn_q_proj/max_abs":5.34375,"train/train/tensor_grad_model_layers_64_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/norm":5.5625,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/max_abs":0.00022029876708984375,"train/train/tensor_act_model_layers_38_mlp/max_abs":0.3984375,"train/train/layer_model_layers_62/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_mlp_gate_proj_weight/norm":7.125,"train/train/tensor_param_model_layers_25_mlp_gate_proj_weight/norm":4.84375,"train/train/tensor_act_model_layers_62_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_up_proj/std":0.35351569571876346,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_mlp_up_proj/max_abs":2.15625,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/max_abs":0.2177734375,"train/train/tensor_act_model_layers_49_self_attn_o_proj/norm":595.4883635802908,"train/train/tensor_param_model_layers_83_mlp_gate_proj_weight/max_abs":0.2060546875,"train/train/tensor_act_model_layers_35_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/mean":-6.324262358248234e-08,"train/train/tensor_act_model_layers_4/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp_gate_proj/mean":-0.0115966796875,"train/train/tensor_act_model_layers_70_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/std":0.0390625,"train/train/tensor_act_model_layers_9/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn/mean":-0.0011615753173828125,"train/train/tensor_act_model_layers_24_mlp/norm":213.7213267493991,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/mean":0.0003452301025390625,"train/train/tensor_act_model_layers_85_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/max_abs":0.291015625,"train/train/tensor_act_model_layers_48_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_k_proj/norm":5086.576597856501,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/max_abs":0.1669921875,"train/train/tensor_act_model_layers_58/frac_near_user_limit":0,"train/train/layer__model_layers_15/param/mean":0.0014836375315364176,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/mean":-4.673004150390625e-05,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/max_abs":0.2578125,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/max_abs":0.000316619873046875,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/max_abs":0.00164031982421875,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/std":0.0233154296875,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/max_abs":0.00016307830810546875,"train/train/tensor_act_model_layers_39_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/std":2.7726268189040703e-05,"train/train/layer_model_layers_26/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp_up_proj/std":0.2509786314916121,"train/train/tensor_act_model_layers_86_mlp/norm":1565.8012737696185,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp_gate_proj/max_abs":2.625,"train/train/tensor_act_model_layers_42/mean":-0.017578125,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/norm":5.15625,"train/train/tensor_act_model_layers_59_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/mean":-0.000202178955078125,"train/train/tensor_act_model_layers_29_mlp/mean":0.0004429817199707031,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/norm":0.03633740214363204,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/max_abs":0.12890625,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/mean":0.0004177093505859375,"train/train/layer_model_layers_42/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74/norm":8960.149419491248,"train/train/tensor_grad_model_layers_31_mlp_gate_proj_weight/norm":0.016181339682414867,"train/train/layer__model_layers_59/param/mean":0.0015570630149424726,"train/train/layer_model_layers_84/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn/max_abs":2.96875,"train/train/tensor_act_model_layers_30_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_post_attention_layernorm/max_abs":5.53125,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/mean":-1.548323780298233e-08,"train/train/tensor_param_model_layers_60_mlp_gate_proj_weight/std":0.03564453125,"train/train/tensor_act_model_layers_88_self_attn_q_proj/mean":0.03875732421875,"train/train/tensor_act_model_layers_16_self_attn_k_proj/max_abs":6.5,"train/train/tensor_act_model_layers_93_self_attn_k_proj/max_abs":5.21875,"train/train/tensor_act_model_layers_10_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/max_abs":0.00017261505126953125,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/norm":0.012916472559157456,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/mean":-1.329183578491211e-05,"train/train/tensor_act_model_layers_63_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_66/grad/max_abs":0.0018157958984375,"train/train/layer_model_layers_36/grad/mean":4.7420083244393656e-08,"train/train/tensor_param_model_layers_13_mlp_gate_proj_weight/mean":-3.147125244140625e-05,"train/train/tensor_act_model_layers_40_self_attn/norm":534.4974967848466,"train/train/tensor_act_model_layers_60_self_attn_q_proj/max_abs":9.6875,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/std":0.0001012161245709067,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/mean":1.9668368622660637e-07,"train/train/tensor_act_model_layers_49_mlp_down_proj/std":0.0683593814926485,"train/train/tensor_grad_model_layers_2_mlp_gate_proj_weight/norm":0.032425756107270484,"train/train/tensor_act_model_layers_7_self_attn_o_proj/max_abs":0.55859375,"train/train/tensor_act_model_layers_25_self_attn_o_proj/mean":0.0005245208740234375,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/std":9.932295750777586e-05,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/norm":5.8125,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/max_abs":0.00164794921875,"train/train/tensor_act_model_layers_21_mlp_up_proj/max_abs":2.109375,"train/train/tensor_act_model_layers_50_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_post_attention_layernorm/mean":-0.0089569091796875,"train/train/tensor_act_model_layers_59_self_attn_k_proj/std":0.8632814290836199,"train/train/layer_model_layers_43/grad/norm":0.03973581297332046,"train/train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/mean":-0.0001392364501953125,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/norm":0.030130664961554813,"train/train/tensor_act_model_layers_82_mlp/mean":0.0010061264038085938,"train/train/layer_model_layers_51/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/max_abs":6.28125,"train/train/tensor_act_model_layers_58_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn/max_abs":4.75,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/norm":0.01180028367216794,"train/train/tensor_act_model_layers_75_mlp_gate_proj/std":0.4746095052471689,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/norm":5,"train/train/tensor_act_model_layers_32_mlp/std":0.04486098303452571,"train/train/layer_model_layers_56/grad/norm":0.05188170878618982,"train/train/tensor_act_model_layers_55_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_k_proj/std":0.9638727791784878,"train/train/tensor_act_model_layers_82/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24/norm":7300.026493013311,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/std":6.48632071562773e-05,"train/train/tensor_act_model_layers_18_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_mlp_gate_proj_weight/mean":2.8233625926077366e-07,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/mean":8.364440873265266e-08,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp_gate_proj/norm":4281.3660109291095,"train/train/tensor_act_model_layers_35_self_attn_o_proj/max_abs":0.6953125,"train/train/tensor_act_model_layers_80_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/norm":0.01612437473238122,"train/train/layer__model_layers_45/param/max_abs":1,"train/train/tensor_act_model_layers_3_mlp_down_proj/max_abs":1.4296875,"train/train/layer__model_layers_50/param/std":0.05379837336588239,"train/train/tensor_act_model_layers_66_self_attn_o_proj/std":0.1997145803489047,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/std":0.0263671875,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/norm":0.008990055858986963,"train/train/tensor_act_model_layers_64/max_abs":9.9375,"train/train/tensor_act_model_layers_89/std":2.2422032897236246,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_12_mlp_gate_proj_weight/norm":0.015738107659013848,"train/train/tensor_act_model_layers_41_mlp_down_proj/max_abs":0.470703125,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/max_abs":0.306640625,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/norm":0.015484776719749661,"train/train/tensor_act_model_layers_34_mlp_gate_proj/norm":2354.5097345624445,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/norm":0.00290376399602362,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/std":0.0274658203125,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/max_abs":0.0013885498046875,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/max_abs":0.000881195068359375,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/norm":5.5625,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/max_abs":0.00131988525390625,"train/train/tensor_act_model_layers_71_mlp_down_proj/std":0.1379401859964465,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/mean":0.00025177001953125,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_11_mlp_up_proj/mean":0.0005189180374145508,"train/train/tensor_act_model_layers_34_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/mean":-2.235756255686283e-07,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/mean":1.0310031939297915e-08,"train/train/layer_model_layers_1/act/norm":13295.389039201495,"train/train/tensor_act_model_layers_56_self_attn_q_proj/mean":-0.0089569091796875,"train/train/tensor_act_model_layers_9_self_attn_k_proj/std":0.8408220935762795,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/max_abs":0.00020122528076171875,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/std":6.912470115575724e-05,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/std":8.206585587319306e-05,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_v_proj/std":0.34179688289761534,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/norm":4.71875,"train/train/tensor_act_model_layers_17_self_attn_o_proj/max_abs":0.734375,"train/train/tensor_act_model_layers_48_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/max_abs":0.000537872314453125,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_1_mlp_gate_proj_weight/mean":2.561137080192566e-08,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/mean":6.246566772460938e-05,"train/train/layer__model_layers_48/param/mean":0.0016292089977056113,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/mean":-4.873145371675491e-07,"train/train/tensor_act_model_layers_77_input_layernorm/norm":5792.608520512642,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/max_abs":7.62939453125e-05,"train/train/tensor_act_model_layers_40_mlp_down_proj/std":0.05706797547549185,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/std":0.04736328125,"train/train/tensor_param_model_layers_58_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/mean":5.103647708892822e-06,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/mean":-0.000804901123046875,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/std":2.887610967806374e-05,"train/train/tensor_act_model_layers_87_mlp_gate_proj/max_abs":3.453125,"train/train/tensor_act_model_layers_56_self_attn_v_proj/norm":2224.9893699124614,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_k_proj/std":0.6933626094259121,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/mean":2.353917807340622e-07,"train/train/layer_model_layers_47/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_post_attention_layernorm/std":1.0000002550877656,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/norm":0.0014284500527920463,"train/train/tensor_act_model_layers_77_self_attn_k_proj/std":1.0566463558443528,"train/train/tensor_act_model_layers_16_mlp_up_proj/max_abs":1.7265625,"train/train/tensor_act_model_layers_52_input_layernorm/max_abs":6.03125,"train/train/tensor_param_model_layers_29_mlp_gate_proj_weight/max_abs":0.1259765625,"train/train/layer_model_layers_1/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_k_proj/mean":-0.02069091796875,"train/train/layer_model_layers_59/grad/mean":-4.0907000137751635e-08,"train/train/tensor_grad_model_layers_15_mlp_gate_proj_weight/max_abs":0.0004749298095703125,"train/train/tensor_act_model_layers_32/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/mean":-0.0003204345703125,"train/train/tensor_grad_model_layers_82_mlp_gate_proj_weight/mean":-5.778856575489044e-07,"train/train/tensor_param_model_layers_61_mlp_gate_proj_weight/std":0.03662109375,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/norm":7.375,"train/train/tensor_act_model_layers_12_self_attn_k_proj/norm":5064.793738711538,"train/train/tensor_act_model_layers_78_mlp_down_proj/std":0.1784673513232102,"train/train/layer_model_layers_14/act/max_abs":8.3125,"train/train/tensor_act_model_layers_13_self_attn_o_proj/mean":7.62939453125e-05,"train/train/layer_model_layers_81/grad/max_abs":0.0011749267578125,"train/train/tensor_act_model_layers_78_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/mean":-2.241540641989559e-08,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/std":0.037353515625,"train/train/tensor_act_model_layers_34_post_attention_layernorm/max_abs":6.6875,"train/train/tensor_act_model_layers_53_self_attn_k_proj/std":0.7998065104248363,"train/train/tensor_act_model_layers_76_self_attn_k_proj/mean":-0.00669097900390625,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_93_self_attn/mean":-0.0043792724609375,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/frac_near_dtype_limit":0,"train_steps_per_second":0.446,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/mean":-6.05359673500061e-08,"train/train/tensor_act_model_layers_25_input_layernorm/max_abs":6.34375,"train/train/tensor_act_model_layers_5_self_attn_k_proj/std":1.0488335961815403,"train/train/tensor_act_model_layers_61_mlp_up_proj/mean":-0.002811431884765625,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/mean":0.000518798828125,"train/train/tensor_act_model_layers_84/max_abs":11.5,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/norm":0.015279191682032378,"train/train/tensor_param_model_layers_4_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_70_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58/mean":0.00013256072998046875,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/norm":0.0020085495051170204,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/mean":0.0003070831298828125,"train/train/tensor_act_model_layers_85_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/mean":-2.7194619178771973e-07,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_mlp_gate_proj_weight/max_abs":0.1484375,"train/train/tensor_act_model_layers_42_mlp/std":0.05969251289088695,"train/train/tensor_param_model_layers_75_mlp_gate_proj_weight/mean":-9.202957153320312e-05,"train/train/tensor_grad_model_layers_63_mlp_gate_proj_weight/norm":0.026374957034903147,"train/train/tensor_act_model_layers_3_post_attention_layernorm/max_abs":5,"train/train/tensor_act_model_layers_31_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_v_proj/max_abs":2.53125,"train/train/tensor_act_model_layers_81_mlp_up_proj/norm":4236.168020406372,"train/train/tensor_act_model_layers_36_self_attn_o_proj/std":0.1033943937408615,"train/train/tensor_act_model_layers_41_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_mlp_gate_proj_weight/max_abs":0.173828125,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/norm":0.003645453724228691,"train/train/tensor_act_model_layers_64_mlp_up_proj/frac_near_user_limit":0,"train/train/layer_model_layers_79/act/frac_near_dtype_limit":0,"train/train/global/param/std":0.05794901779234425,"train/train/tensor_act_model_layers_47_self_attn_q_proj/max_abs":7.46875,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/norm":6.09375,"train/train/layer__model_layers_25/param/mean":0.0015556763933154982,"train/train/tensor_act_model_layers_19_self_attn_o_proj/max_abs":1.15625,"train/train/tensor_param_model_layers_42_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/mean":-1.816079020500183e-08,"train/train/tensor_act_model_layers_57_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/max_abs":0.2333984375,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/std":5.0211601006772545e-05,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/mean":-0.000499725341796875,"train/train/tensor_grad_model_layers_62_mlp_gate_proj_weight/mean":3.230525180697441e-08,"train/train/tensor_act_model_layers_91_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/max_abs":0.251953125,"train/train/tensor_act_model_layers_6_self_attn_o_proj/std":0.06408932496959693,"train/train/tensor_grad_model_embed_tokens_weight/mean":1.3015232980251312e-07,"train/train/tensor_act_model_layers_74_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/mean":1.0006129741668701e-05,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/mean":2.203509211540222e-06,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/std":0.0001278942142783809,"train/train/layer__model_layers_41/param/frac_near_user_limit":0,"train/train/layer_model_layers_88/grad/std":7.905554095686525e-05,"train/train/tensor_grad_model_layers_48_mlp_gate_proj_weight/norm":0.018506640415266933,"train/train/layer__model_layers_71/param/std":0.05781508159219127,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_post_attention_layernorm/max_abs":6.25,"train/train/tensor_act_model_layers_12/max_abs":8.5,"train/train/tensor_param_model_layers_48_mlp_gate_proj_weight/max_abs":0.1396484375,"train/train/tensor_param_model_layers_43_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn/norm":807.951509744037,"train/train/tensor_act_model_layers_37_mlp_gate_proj/norm":2448.832587211442,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/max_abs":0.0002593994140625,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/max_abs":0.0004596710205078125,"train/train/tensor_act_model_layers_78/std":1.6855525388767445,"train/train/tensor_param_model_layers_62_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/norm":5.59375,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/norm":0.0005753178146266794,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_81_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_39_self_attn_k_proj/mean":-0.02923583984375,"train/train/tensor_param_model_layers_81_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/norm":0.016844376221518364,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_mlp_gate_proj_weight/norm":9.5,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/mean":-0.0002994537353515625,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/norm":7.1875,"train/train/tensor_param_model_layers_28_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_input_layernorm/max_abs":5.125,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/mean":1.4055520296096802e-05,"train/train/tensor_act_model_layers_20_mlp/std":0.03594987744548412,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_down_proj/max_abs":0.3671875,"train/train/layer_model_layers_93/grad/max_abs":0.00115966796875,"train/train/tensor_param_model_layers_28_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/max_abs":0.00064849853515625,"train/train/tensor_act_model_layers_49_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer__model_layers_67/param/max_abs":1,"train/train/tensor_act_model_layers_4_self_attn/mean":0.0016803741455078125,"train/train/tensor_grad_model_layers_3_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/max_abs":0.00110626220703125,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/max_abs":0.00015354156494140625,"train/train/tensor_act_model_layers_1_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_up_proj/mean":0.0023097991943359375,"train/train/tensor_act_model_layers_87_mlp_up_proj/max_abs":3.53125,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/std":1.2838738716519626e-05,"train/train/tensor_param_model_layers_21_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_39_self_attn_o_proj/max_abs":1.0078125,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_v_proj/max_abs":3.515625,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/norm":0.0013224846265522164,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/std":0.036865234375,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/mean":-7.43865966796875e-05,"train/train/tensor_act_model_layers_86_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/mean":0.00010776519775390625,"train/train/tensor_act_model_layers_80_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_69_self_attn_o_proj/mean":-0.0004181861877441406,"train/train/tensor_act_model_layers_86_self_attn_v_proj/max_abs":3.0625,"train/train/tensor_act_model_layers_55_mlp_up_proj/mean":-0.0011882781982421875,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/std":3.0039124967792995e-05,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/mean":0.0004100799560546875,"train/train/tensor_act_model_layers_29/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/norm":0.0013606389348752724,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/norm":0.005128088325890064,"train/train/tensor_act_model_layers_82_mlp_down_proj/std":0.21020563520226931,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/max_abs":0.0013580322265625,"train/train/layer__model_layers_2/param/norm":19.2010716457116,"train/train/tensor_act_model_layers_30_mlp/std":0.04199220052768469,"train/train/tensor_act_model_layers_56_self_attn/norm":642.2776714730744,"train/train/tensor_act_model_layers_50_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_mlp_gate_proj_weight/std":7.547225777825606e-05,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/mean":-2.8371810913085938e-05,"train/train/tensor_act_model_layers_44_self_attn/mean":0.001178741455078125,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/mean":8.177012205123901e-06,"train/train/tensor_grad_model_layers_72_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/norm":0.001834095972320056,"train/train/tensor_act_model_layers_72_self_attn_v_proj/norm":2441.160365564245,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn/mean":0.0003604888916015625,"train/train/tensor_param_model_layers_80_mlp_gate_proj_weight/norm":7.71875,"train/train/tensor_act_model_layers_59_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/norm":4.0625,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn_v_proj/std":0.41943453437824046,"train/train/tensor_act_model_layers_17_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_k_proj/norm":5104.447046001726,"train/train/tensor_act_model_layers_29_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp_gate_proj/mean":0.0007839202880859375,"train/train/tensor_param_model_layers_16_mlp_gate_proj_weight/norm":4.65625,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_77/act/std":0.7855009375903573,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_26_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/max_abs":0.000370025634765625,"train/train/tensor_act_model_layers_4_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn/max_abs":2.46875,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/mean":1.9790604710578918e-07,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/mean":3.032619133591652e-08,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_mlp_gate_proj_weight/mean":-4.842877388000488e-08,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/mean":-5.133915692567825e-08,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/std":7.175559088366568e-05,"train/train/layer__model_layers_30/param/std":0.051338213009775334,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/std":2.6998935487503283e-05,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/norm":7.15625,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/max_abs":0.00022220611572265625,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/std":2.578776109736979e-05,"train/train/tensor_act_model_layers_82/std":1.839851366091108,"train/train/layer_model_layers_72/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/max_abs":0.80859375,"train/train/tensor_act_model_layers_66_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/max_abs":0.123046875,"train/train/tensor_param_model_layers_59_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_58_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_67/act/max_abs":10.875,"train/train/tensor_act_model_layers_75_self_attn_o_proj/max_abs":1.671875,"train/train/tensor_act_model_layers_87/max_abs":11.875,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/std":0.033447265625,"train/train/tensor_param_model_layers_78_mlp_gate_proj_weight/norm":7.5625,"train/train/tensor_act_model_layers_72_self_attn_q_proj/mean":0.0694580078125,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/mean":0.000102996826171875,"train/train/tensor_act_model_layers_35_mlp_gate_proj/std":0.2890625309098395,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/std":2.18063611269582e-05,"train/train/tensor_act_model_layers_8_self_attn_q_proj/max_abs":4.8125,"train/loss":7.199232482910157,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/max_abs":0.000812530517578125,"train/train/tensor_act_model_layers_74_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_post_attention_layernorm/mean":-0.0015716552734375,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/std":0.05126953125,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/std":0.02734375,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/max_abs":0.000499725341796875,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/max_abs":0.000492095947265625,"train/train/layer_model_layers_3/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/mean":-4.00003045797348e-07,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/max_abs":0.0004062652587890625,"train/train/tensor_act_model_layers_82_self_attn_q_proj/std":1.0156250201738797,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/max_abs":0.1337890625,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/mean":0.00015926361083984375,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_62/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_q_proj/mean":0.03753662109375,"train/train/tensor_act_model_layers_16_mlp_up_proj/mean":-0.001262664794921875,"train/train/tensor_act_model_layers_41_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn_k_proj/norm":6416.16611248889,"train/train/tensor_act_model_layers_58_mlp_up_proj/max_abs":2.25,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/norm":5.84375,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/norm":0.01349946715486612,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/std":0.03662109375,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_input_layernorm/max_abs":5.8125,"train/train/tensor_act_model_layers_85_self_attn_v_proj/norm":2535.0357823850522,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/max_abs":0.00063323974609375,"train/train/tensor_act_model_layers_62_input_layernorm/mean":5.626678466796875e-05,"train/train/tensor_act_model_layers_40/max_abs":9.375,"train/train/tensor_act_model_layers_86/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/norm":4.75,"train/train/tensor_act_model_layers_35_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp_down_proj/mean":-0.00180816650390625,"train/train/tensor_param_model_layers_2_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_80_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_post_attention_layernorm/mean":-0.01409912109375,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/max_abs":0.1943359375,"train/train/tensor_act_model_layers_40_mlp/mean":0.0002522468566894531,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_v_proj/mean":-0.0031280517578125,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/max_abs":9.679794311523438e-05,"train/train/tensor_act_model_layers_22_self_attn/max_abs":0.7109375,"train/train/tensor_grad_model_layers_58_mlp_gate_proj_weight/norm":0.023356057763986488,"train/train/tensor_act_model_layers_85_self_attn_q_proj/mean":0.025238037109375,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/max_abs":0.00040435791015625,"train/train/tensor_act_model_layers_58_mlp_gate_proj/std":0.3867187770930194,"train/train/tensor_act_model_layers_1_mlp_down_proj/std":0.2285166778856051,"train/train/tensor_act_model_layers_57_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_input_layernorm/norm":5792.608032230507,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/max_abs":0.2470703125,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/mean":-4.971399903297424e-06,"train/train/layer_model_layers_20/grad/mean":-4.681076436026048e-08,"train/train/tensor_param_model_layers_50_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_90/act/mean":-0.015427930014474052,"train/train/tensor_act_model_layers_49_input_layernorm/norm":5792.612548829056,"train/train/tensor_grad_model_layers_52_mlp_gate_proj_weight/mean":-1.2747477740049362e-08,"train/train/tensor_act_model_layers_16/max_abs":8.4375,"train/train/tensor_param_model_layers_9_mlp_gate_proj_weight/mean":-6.389617919921875e-05,"train/train/tensor_act_model_layers_58_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp/norm":682.7734066122232,"train/train/tensor_act_model_layers_36_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_72/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_v_proj/norm":2182.833831801511,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/max_abs":0.004974365234375,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/mean":-0.0002841949462890625,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_mlp_down_proj/norm":699.4997944017726,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/max_abs":0.201171875,"train/train/tensor_act_model_layers_24_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_lm_head/mean":-2.23828125,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_9/param/max_abs":1,"train/train/tensor_act_model_layers_79_self_attn_v_proj/std":0.40625003152168593,"train/train/layer_model_layers_80/act/max_abs":11.375,"train/train/tensor_grad_model_layers_91_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/norm":4.625,"train/train/tensor_act_model_layers_4_self_attn_v_proj/mean":-0.00542449951171875,"train/train/tensor_act_model_layers_37_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/max_abs":0.2421875,"train/train/tensor_act_model_layers_35_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_k_proj/std":0.9658271627308572,"train/train/layer_model_layers_57/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/std":2.8897182979969126e-05,"train/train/tensor_param_model_layers_33_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_55/grad/mean":1.1794815383351723e-08,"train/train/tensor_act_model_layers_1_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81/std":1.8007886017847903,"train/train/tensor_act_model_layers_57_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/std":0.05615234375,"train/train/tensor_param_model_layers_51_input_layernorm_weight/mean":1,"train/train/layer_model_layers_92/grad/mean":9.229828061626035e-08,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/mean":-0.000171661376953125,"train/train/tensor_act_model_layers_89_self_attn_v_proj/max_abs":2.84375,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_gate_proj/mean":0.0097503662109375,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/mean":-0.000125885009765625,"train/train/tensor_act_model_layers_46_post_attention_layernorm/norm":5792.607666024788,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/norm":7.8125,"train/train/tensor_param_model_layers_78_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/max_abs":0.0003948211669921875,"train/train/tensor_act_model_layers_50_self_attn_k_proj/std":0.7998066171161724,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/mean":-0.00010061264038085938,"train/train/tensor_act_model_layers_3_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/max_abs":0.158203125,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/norm":0.02370671783054035,"train/train/tensor_act_model_layers_36/std":1.2421885191827768,"train/train/tensor_act_model_layers_4/mean":-0.0318603515625,"train/train/tensor_act_model_layers_42_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer__model_layers_44/param/mean":0.0014798882970943838,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/norm":3.828125,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65/mean":0.002773284912109375,"train/train/tensor_grad_model_layers_73_mlp_gate_proj_weight/norm":0.02731873867005877,"train/train/tensor_act_model_layers_60/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/std":1.062972770099644e-05,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_up_proj/norm":2014.9239446649335,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/std":9.216485708419647e-05,"train/train/tensor_act_model_layers_77/norm":9582.02030464272,"train/train/tensor_act_model_layers_6_input_layernorm/max_abs":5.125,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/std":2.0153262957126327e-05,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/norm":0.01673120480382755,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/std":0.00010684451448546162,"train/train/tensor_act_model_layers_29_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/norm":0.014305845114818517,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/max_abs":0.1328125,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn/frac_near_user_limit":0,"total_flos":3.6392499412992e+16,"train/train/tensor_act_model_layers_86_mlp_up_proj/mean":0.00374603271484375,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/std":6.905268450387742e-05,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/mean":0.000202178955078125,"train/train/global/act/frac_near_user_limit":0,"train/train/layer__model_layers_87/param/norm":25.13543393423913,"train/train/tensor_grad_model_layers_91_mlp_gate_proj_weight/mean":9.424984455108643e-07,"train/train/layer_model_layers_25/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_k_proj/mean":-0.0016565322875976562,"train/train/tensor_act_model_layers_26_mlp_down_proj/norm":225.82528829992927,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/std":0.0242919921875,"train/train/tensor_act_model_layers_54_mlp/norm":468.6209762802814,"train/train/tensor_act_model_layers_0_self_attn_o_proj/max_abs":0.283203125,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/std":4.8262247421262894e-05,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/std":4.677300971879779e-05,"train/train/tensor_act_model_layers_39_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_18/grad/mean":-4.536892155217306e-08,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/mean":0.00011110305786132812,"train/train/tensor_act_model_layers_44/max_abs":9.375,"train/train/tensor_act_model_layers_46_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/norm":8.6875,"train/train/tensor_act_model_layers_83_self_attn/mean":-6.103515625e-05,"train/train/tensor_act_model_layers_6_mlp_up_proj/norm":1933.5794128031016,"train/train/layer_model_layers_64/grad/max_abs":0.00183868408203125,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/std":0.0220947265625,"train/train/tensor_act_model_norm/mean":0.0063323974609375,"train/train/tensor_act_model_layers_20/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/std":8.09053692800423e-05,"train/train/tensor_grad_model_layers_77_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_29/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_up_proj/mean":0.00492095947265625,"train/train/tensor_act_model_layers_2_mlp_up_proj/mean":0.0155181884765625,"train/train/tensor_act_model_layers_84_self_attn_q_proj/std":1.2441456314058048,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_q_proj/std":0.9853574357552434,"train/train/tensor_act_model_layers_78_mlp/mean":0.0010099411010742188,"train/train/tensor_act_model_layers_47_self_attn_q_proj/std":0.9384863030020459,"train/train/tensor_act_model_layers_30_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_post_attention_layernorm/std":1.0000008981074249,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/max_abs":0.00157928466796875,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/mean":-0.0003566741943359375,"train/train/tensor_act_model_layers_5_self_attn_k_proj/mean":-0.026519775390625,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/mean":4.738103598356247e-07,"train/train/tensor_act_model_layers_0_mlp/mean":-0.0223388671875,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/max_abs":0.000732421875,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/norm":6.875,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn/mean":0.00015664100646972656,"train/train/tensor_act_model_layers_78_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn/max_abs":2.296875,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn_k_proj/std":0.779299673873732,"train/train/tensor_act_model_layers_16/norm":7411.068636023288,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/max_abs":0.14453125,"train/train/tensor_act_model_layers_93_self_attn_o_proj/mean":-0.0043792724609375,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_65_mlp_up_proj/frac_near_user_limit":0,"train/train/layer_model_layers_65/act/norm":15142.480750519762,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/max_abs":0.208984375,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/norm":8,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/norm":5,"train/train/tensor_param_model_layers_24_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22/max_abs":8.5625,"train/train/tensor_grad_model_layers_12_mlp_gate_proj_weight/max_abs":0.0004634857177734375,"train/train/tensor_act_model_layers_38/std":1.2480522122627182,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/mean":1.3748649507761002e-07,"train/train/tensor_grad_model_layers_50_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/mean":0.0001544952392578125,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/mean":0.0008697509765625,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/mean":-0.00058746337890625,"train/train/tensor_act_model_layers_29_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn_k_proj/std":0.7783223873670772,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/max_abs":0.000560760498046875,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/norm":6.4375,"train/train/layer_model_layers_34/grad/max_abs":0.00128936767578125,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/max_abs":0.1513671875,"train/train/tensor_act_model_layers_7/norm":7476.470182956397,"train/train/tensor_act_model_layers_9_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/std":0.00011651756462392147,"train/train/layer_model_layers_89/grad/mean":6.3728035519164e-08,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/max_abs":0.15234375,"train/train/tensor_act_model_layers_29_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_mlp_gate_proj_weight/mean":1.0512303560972214e-07,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/max_abs":0.0008087158203125,"train/train/tensor_act_model_layers_84_mlp_up_proj/max_abs":3.03125,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/std":0.00010475320546162324,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/norm":0.0007507058658399955,"train/train/tensor_act_model_layers_49_post_attention_layernorm/norm":5792.610107423568,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/std":4.315685513602136e-05,"train/train/tensor_act_model_layers_16_self_attn_q_proj/max_abs":7.21875,"train/train/tensor_act_model_layers_22_input_layernorm/max_abs":6.3125,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_4_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn_q_proj/max_abs":7.34375,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/std":7.439180182835033e-05,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/max_abs":0.16015625,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/std":2.7977445808975212e-05,"train/train/tensor_grad_model_layers_67_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92/frac_near_user_limit":0,"train/train/layer_model_layers_47/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_gate_proj_weight/norm":0.126708984375,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/mean":8.493661880493164e-07,"train/train/tensor_act_model_layers_41_input_layernorm/mean":-0.0134735107421875,"train/train/tensor_act_model_layers_87_self_attn_v_proj/std":0.46728603997989027,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/std":0.0252685546875,"train/train/layer__model_layers_24/param/norm":19.915262382937616,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp_down_proj/std":0.06048593385180993,"train/train/tensor_act_model_layers_47_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_mlp_gate_proj/std":0.2812500310440841,"train/train/layer_model_layers_55/act/mean":0.003603083746773856,"train/train/layer_model_layers_76/act/max_abs":11,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/std":4.0025481503039545e-05,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/norm":0.0009901325814935466,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/std":5.7175338101645124e-05,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/std":0.040283203125,"train/train/tensor_act_model_layers_51_mlp_up_proj/std":0.3535156792533949,"train/train/tensor_act_model_layers_54_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/max_abs":0.000701904296875,"train/train/tensor_act_model_layers_61_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/mean":1.6883015632629395e-05,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_q_proj/norm":5414.788625963886,"train/train/tensor_param_model_layers_19_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_o_proj/std":0.08496094085697624,"train/train/tensor_act_model_layers_89_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71/mean":0.005017280578613281,"train/train/tensor_act_model_layers_2/std":1.2949265786340134,"train/train/tensor_act_model_layers_75_self_attn_q_proj/norm":6201.253604544134,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/std":0.04443359375,"train/train/tensor_act_model_layers_31_mlp_down_proj/mean":0.001064300537109375,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/std":0.0007432123514805369,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn/std":0.07629508940315724,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/norm":8.0625,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/norm":0.021905839842739982,"train/train/tensor_act_model_layers_64_self_attn_q_proj/norm":5268.08835911496,"train/train/tensor_param_model_layers_19_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_24_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_91/act/max_abs":14.5,"train/train/tensor_act_model_layers_27_self_attn_k_proj/mean":-0.0025119781494140625,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/std":0.044677734375,"train/train/tensor_param_model_layers_78_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/norm":0.014031881536956553,"train/train/tensor_act_model_layers_34_mlp_down_proj/std":0.047058246208170944,"train/train/tensor_grad_model_layers_88_mlp_gate_proj_weight/mean":2.2136373445391655e-07,"train/train/tensor_param_model_layers_20_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_84/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/norm":4.25,"train/epoch":0.8088433540037746,"train/train/tensor_grad_model_layers_21_mlp_gate_proj_weight/std":4.108684546205441e-05,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/global/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_input_layernorm/max_abs":5.90625,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/mean":3.4761615097522736e-07,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/norm":0.0008860429947759785,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/norm":0.022529662765488293,"train/train/tensor_act_model_layers_6_self_attn_v_proj/max_abs":1.6875,"train/train/tensor_act_model_layers_84_self_attn_o_proj/max_abs":2.46875,"train/train/layer__model_layers_67/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_v_proj/std":0.37695315658315875,"train/train/tensor_act_model_layers_34_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/max_abs":0.1259765625,"train/train/tensor_act_model_layers_56_post_attention_layernorm/max_abs":5.875,"train/train/tensor_param_model_layers_65_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_input_layernorm/std":1.0000003217718978,"train/train/layer_model_layers_41/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/max_abs":0.2490234375,"train/train/tensor_act_model_layers_44_mlp_gate_proj/norm":2697.589824563774,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/norm":0.023901713196351152,"train/train/tensor_grad_model_layers_48_mlp_gate_proj_weight/std":5.117818924631519e-05,"train/train/tensor_act_model_layers_27/mean":-0.01947021484375,"train/train/layer__model_layers_28/param/mean":0.001458208200154178,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_59_self_attn_o_proj/mean":0.00028634071350097656,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/mean":5.471520125865936e-09,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/std":6.183528036006784e-05,"train/train/tensor_act_model_layers_10_self_attn/mean":0.0006742477416992188,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/std":0.05517578125,"train/train/tensor_param_model_layers_43_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/max_abs":0.0004787445068359375,"train/train/tensor_act_model_layers_26_self_attn_v_proj/max_abs":2.34375,"train/train/tensor_param_model_layers_43_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/mean":-2.6007910491898656e-08,"train/train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp_down_proj/max_abs":1.2734375,"train/train/tensor_act_model_layers_91_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn/max_abs":1.5703125,"train/train/layer_model_layers_84/grad/norm":0.06777010581047666,"train/train/tensor_act_model_layers_86_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/max_abs":0.000308990478515625,"train/train/tensor_act_model_layers_49/std":1.2597707415632093,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/std":0.0001417813805633599,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/std":0.0361328125,"train/train/tensor_act_model_layers_41_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_k_proj/mean":0.018157958984375,"train/train/tensor_act_model_layers_89_self_attn_q_proj/mean":0.0977783203125,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/std":0.0238037109375,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/std":8.319778913618643e-05,"train/train/tensor_act_model_layers_30_input_layernorm/mean":-0.0141448974609375,"train/train/tensor_act_model_layers_71_input_layernorm/max_abs":5.75,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/std":0.00012362726912094532,"train/train/tensor_param_model_layers_40_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_29/grad/max_abs":0.00164794921875,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp_up_proj/std":0.36914064013768727,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/max_abs":0.00015163421630859375,"train/train/tensor_act_model_layers_5_mlp_up_proj/norm":1874.4079692302998,"train/train/layer_model_layers_13/grad/norm":0.04018898139881122,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/std":0.046142578125,"train/train/tensor_act_model_layers_29_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_16/act/mean":-0.004471336092267718,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/std":0.052001953125,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_k_proj/mean":0.01025390625,"train/train/tensor_act_model_layers_22_self_attn_v_proj/norm":1795.7601173601508,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/mean":1.256412360817194e-07,"train/train/tensor_act_model_layers_32_self_attn_o_proj/max_abs":1.078125,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/norm":0.014808605651931185,"train/train/tensor_act_model_layers_53_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_up_proj/norm":1997.5704623264876,"train/train/layer_model_layers_70/act/std":0.7073992437769716,"train/train/tensor_grad_model_layers_90_mlp_gate_proj_weight/norm":0.033400677165288016,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/norm":0.0020206844260480506,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_mlp_gate_proj_weight/max_abs":0.14453125,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61/std":1.3242253648834483,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/mean":8.732080459594727e-06,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_mlp_gate_proj_weight/norm":0.018286218375602976,"train/train/tensor_grad_model_layers_47_mlp_gate_proj_weight/max_abs":0.00054931640625,"train/train/tensor_act_model_layers_89_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp/std":0.08105471055584082,"train/train/tensor_act_model_layers_58_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/std":1.6375007465037264e-05,"train/train/layer__model_layers_85/param/std":0.060645484226058116,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_q_proj/mean":0.0325927734375,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/norm":0.0012698874524055893,"train/train/tensor_act_model_layers_49_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_k_proj/max_abs":5.15625,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/max_abs":0.00106048583984375,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/std":4.323870413473516e-05,"train/train/tensor_act_model_layers_40_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/mean":-2.1257437765598297e-07,"train/train/tensor_act_model_layers_63_self_attn_v_proj/mean":0.00347900390625,"train/train/tensor_grad_model_layers_82_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/norm":6.34375,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/mean":2.1420419216156006e-08,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/std":0.02392578125,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_v_proj/norm":2576.9615672538257,"train/train/tensor_act_model_layers_50_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_gate_proj_weight/max_abs":0.000896453857421875,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_mlp_down_proj/std":0.03875747248795357,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/std":0.0419921875,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_gate_proj_weight/max_abs":0.000621795654296875,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/max_abs":0.0005035400390625,"train/train/tensor_act_model_layers_49_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn/norm":390.63338948877305,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/max_abs":0.00115966796875,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/norm":6.96875,"train/train/tensor_grad_model_layers_40_mlp_gate_proj_weight/max_abs":0.000537872314453125,"train/train/tensor_param_model_layers_70_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/norm":4.65625,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/norm":4.53125,"train/train/tensor_act_model_layers_18_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_k_proj/mean":0.014617919921875,"train/train/tensor_act_model_layers_74_self_attn_o_proj/mean":-5.924701690673828e-05,"train/train/tensor_act_model_layers_84_input_layernorm/norm":5792.617187506205,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/std":2.5604590020067333e-05,"train/train/tensor_act_model_layers_53_mlp_gate_proj/max_abs":2.140625,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/mean":-0.0003299713134765625,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_60/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29/max_abs":9,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/std":0.0242919921875,"train/train/tensor_act_model_layers_24_mlp/mean":0.0008678436279296875,"train/train/tensor_act_model_layers_55_self_attn_k_proj/std":0.7666037662741699,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/max_abs":0.00067901611328125,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/norm":5.4375,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_gate_proj_weight/mean":-5.210749804973602e-07,"train/train/tensor_act_model_layers_15_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/max_abs":0.00051116943359375,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_52_self_attn_q_proj/mean":-0.01947021484375,"train/train/tensor_act_model_layers_21_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_o_proj/norm":1462.8881707193352,"train/train/tensor_act_model_layers_15_mlp_down_proj/std":0.03973478515877061,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/max_abs":0.0003261566162109375,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/max_abs":0.1259765625,"train/train/layer__model_layers_23/param/std":0.04928731592886734,"train/train/tensor_grad_model_layers_79_mlp_gate_proj_weight/mean":9.19681042432785e-08,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_23/param/norm":19.970009252157347,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/max_abs":0.0012969970703125,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/mean":0.000514984130859375,"train/train/tensor_act_model_layers_17_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/norm":4.3125,"train/train/tensor_param_model_layers_49_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_68/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_o_proj/mean":0.00200653076171875,"train/train/layer_model_layers_40/act/mean":-0.002145665032523019,"train/train/layer_model_layers_4/act/norm":15375.339456894259,"train/train/tensor_act_model_layers_46_mlp_gate_proj/std":0.33007830795799353,"train/train/tensor_act_model_layers_52_self_attn_v_proj/norm":1788.8241183219466,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/mean":-5.1975250244140625e-05,"train/train/tensor_act_model_layers_63_self_attn_v_proj/std":0.36523439332763097,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_84/grad/max_abs":0.0012664794921875,"train/train/tensor_act_model_layers_82_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_down_proj/std":1.2714891243185245,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/std":5.0214857760957e-05,"train/train/tensor_act_model_layers_2_post_attention_layernorm/std":1.0000001317821356,"train/train/layer_model_layers_15/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer_model_layers_92/act/max_abs":18.125,"train/train/tensor_act_model_layers_9_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/mean":2.4872133508324623e-07,"train/train/layer_model_layers_66/grad/mean":-1.3090821212613825e-09,"train/train/tensor_grad_model_layers_62_mlp_gate_proj_weight/std":6.903285351120239e-05,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_post_attention_layernorm/norm":5792.603881839316,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_0_self_attn_k_proj/mean":0.0089263916015625,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/norm":0.03342475969261003,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/mean":0.000396728515625,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/norm":0.002877212495241575,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/mean":-0.00032806396484375,"train/train/tensor_act_model_layers_51_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_53/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/norm":7.875,"train/train/tensor_act_model_layers_87_mlp_down_proj/mean":-0.003604888916015625,"train/train/tensor_act_model_layers_67_self_attn_o_proj/mean":0.0004253387451171875,"train/train/tensor_param_model_layers_21_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_68_input_layernorm/norm":5792.606445312737,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/std":0.028076171875,"train/train/tensor_act_model_layers_64_self_attn_v_proj/mean":-0.0008020401000976562,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/max_abs":0.0002231597900390625,"train/train/tensor_act_model_layers_11_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/max_abs":0.0003795623779296875,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/mean":4.100799560546875e-05,"train/train/layer_model_layers_56/act/norm":14348.427221086069,"train/train/tensor_act_model_layers_74_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/std":1.4570199689834365e-05,"train/train/tensor_param_model_layers_81_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_57_input_layernorm/max_abs":5.84375,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/max_abs":0.001312255859375,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp/max_abs":1.4296875,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/mean":6.088521331548691e-08,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_post_attention_layernorm/std":1.0000008818165702,"train/train/tensor_act_model_layers_75_self_attn_v_proj/mean":0.00496673583984375,"train/train/layer__model_layers_1/param/std":0.04615734756642173,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/std":6.683841729321847e-05,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/mean":-0.00013828277587890625,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/norm":4.8125,"train/train/tensor_param_model_layers_73_mlp_gate_proj_weight/std":0.039306640625,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61/frac_near_dtype_limit":0,"train/train/layer_model_layers_89/grad/std":8.177187367109473e-05,"train/train/tensor_act_model_layers_67_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/std":2.2300723743751057e-05,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/std":4.162370551535219e-05,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21/norm":7341.925228723216,"train/train/tensor_act_model_layers_88_mlp_down_proj/norm":1882.608534771745,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_21/grad/std":4.818611525868619e-05,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/max_abs":0.1279296875,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_16/act/max_abs":8.4375,"train/train/layer__model_layers_41/param/mean":0.0015870806951418682,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_input_layernorm/mean":-0.0111541748046875,"train/train/tensor_grad_model_layers_64_mlp_gate_proj_weight/norm":0.025728834216998384,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/norm":0.0014175745587555045,"train/train/tensor_param_model_layers_11_mlp_gate_proj_weight/std":0.025390625,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/max_abs":0.15234375,"train/train/tensor_act_model_layers_66_mlp_gate_proj/mean":0.0100250244140625,"train/train/tensor_act_model_layers_86_self_attn_o_proj/std":0.2241261923168629,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/mean":1.0617077350616455e-07,"train/train/layer__model_layers_58/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/mean":-6.239861249923706e-07,"train/train/layer_model_layers_37/act/mean":-0.007235799516950335,"train/train/tensor_act_model_layers_55_post_attention_layernorm/norm":5792.6076660225135,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/std":0.049072265625,"train/train/tensor_act_model_layers_70_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_43/act/std":0.6328385254708705,"train/train/tensor_act_model_layers_48_self_attn_q_proj/std":0.8798845709675714,"train/train/layer_model_layers_36/act/std":0.6470009609264149,"train/train/tensor_act_model_layers_85_self_attn/std":0.19165096657774267,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/norm":0.019354401067238926,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/max_abs":0.0002460479736328125,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/norm":7.3125,"train/train/layer_model_layers_19/grad/max_abs":0.00128173828125,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/norm":6.15625,"train/train/tensor_act_model_layers_42_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_o_proj/norm":484.1207608406099,"train/train/tensor_act_model_layers_68/mean":0.008617401123046875,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/std":6.518882072633677e-05,"train/train/tensor_act_model_layers_91_mlp_up_proj/std":0.7031250549687258,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/norm":4.84375,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/std":0.035400390625,"train/train/layer_model_layers_77/act/max_abs":11.125,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/max_abs":0.0009613037109375,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_mlp_gate_proj/max_abs":3.34375,"train/train/tensor_param_model_layers_7_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_28_post_attention_layernorm/norm":5792.613281258561,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/norm":4.9375,"train/train/tensor_grad_model_layers_70_mlp_gate_proj_weight/mean":5.012843757867813e-07,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/max_abs":0.25390625,"train/train/tensor_act_model_layers_40_self_attn_v_proj/mean":-0.0015773773193359375,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/std":5.785377808748446e-05,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/norm":0.026095009847596377,"train/train/layer_model_layers_73/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/max_abs":0.000301361083984375,"train/train/tensor_act_model_layers_93_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_up_proj/mean":-0.00548553466796875,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_27_mlp_gate_proj/mean":-0.003322601318359375,"train/train/tensor_act_model_layers_15_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/norm":0.008529431061838927,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/mean":1.3969838619232178e-09,"train/train/tensor_grad_model_layers_84_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/max_abs":0.001556396484375,"train/train/tensor_act_model_layers_74_mlp_up_proj/norm":3789.8317315704,"train/train/tensor_act_model_layers_77_post_attention_layernorm/std":1.000001785582069,"train/train/tensor_act_model_layers_43_input_layernorm/norm":5792.608764651055,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/mean":5.9604644775390625e-06,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/max_abs":0.11572265625,"train/train/tensor_act_model_layers_19_mlp_gate_proj/std":0.24316408365486045,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/std":4.890776250909837e-05,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/max_abs":0.1591796875,"train/train/tensor_act_model_layers_43_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer__model_layers_68/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_mlp_gate_proj_weight/norm":7.03125,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/mean":5.3551048040390015e-09,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_89/param/std":0.062263963785115874,"train/train/layer_model_layers_81/act/std":0.8262048767587358,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/max_abs":0.1318359375,"train/train/tensor_param_model_layers_47_input_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_73/param/norm":23.436041621293473,"train/train/tensor_grad_model_layers_86_mlp_gate_proj_weight/norm":0.029911336810988638,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/max_abs":0.1806640625,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_k_proj/max_abs":5.65625,"train/train/tensor_param_model_layers_77_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_33_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/std":0.035888671875,"train/train/tensor_act_model_layers_14_mlp_gate_proj/max_abs":1.890625,"train/train/layer_model_layers_13/act/max_abs":8.4375,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/max_abs":0.0004062652587890625,"train/train/tensor_act_model_layers_63_mlp/norm":617.7870798821212,"train/train/tensor_param_model_layers_79_mlp_gate_proj_weight/std":0.0419921875,"train/train/tensor_act_model_layers_71_input_layernorm/std":1.000001844337139,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/max_abs":0.0022735595703125,"train/train/tensor_param_model_layers_10_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/norm":8.1875,"train/train/tensor_act_model_layers_65_mlp_up_proj/mean":-0.009063720703125,"train/train/layer_model_layers_88/act/std":0.8793354352655735,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/std":1.871209604073176e-05,"train/train/tensor_act_model_layers_70_mlp_down_proj/mean":-0.0005512237548828125,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_k_proj/norm":4519.358481450617,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/mean":0.00010585784912109375,"train/train/tensor_grad_model_layers_26_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_34/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/max_abs":0.00145721435546875,"train/train/tensor_act_model_layers_7_mlp_down_proj/norm":246.29436084516988,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_k_proj/std":1.1113335956244415,"train/train/tensor_act_model_layers_46_post_attention_layernorm/mean":-0.011199951171875,"train/train/layer_model_layers_86/grad/mean":2.2757106842749205e-07,"train/train/tensor_act_model_norm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_o_proj/mean":-0.0004475116729736328,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/std":0.0283203125,"train/train/tensor_param_model_layers_93_mlp_gate_proj_weight/norm":10.6875,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/max_abs":0.1875,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/mean":-3.943569026887417e-08,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/max_abs":0.000518798828125,"train/train/tensor_param_model_layers_43_mlp_gate_proj_weight/mean":1.6689300537109375e-05,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/mean":-2.871965989470482e-07,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/std":0.0390625,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/norm":7.15625,"train/train/tensor_grad_model_embed_tokens_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/max_abs":0.000213623046875,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/max_abs":0.11572265625,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_down_proj/norm":230.98925763157234,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/max_abs":9.775161743164062e-05,"train/train/tensor_act_model_layers_16_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/max_abs":0.0021820068359375,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/std":0.0419921875,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/mean":5.245208740234375e-05,"train/train/tensor_act_model_layers_18_mlp/norm":219.4234072503138,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/norm":0.012346199191036111,"train/train/layer_model_layers_30/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80/norm":10167.337995366865,"train/train/tensor_param_model_layers_87_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/std":9.227887797949842e-06,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/max_abs":0.2119140625,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/std":3.910612498571369e-05,"train/train/tensor_param_model_layers_52_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_mlp_down_proj/mean":0.00040435791015625,"train/train/layer__model_layers_72/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/mean":-1.150369644165039e-05,"train/train/tensor_act_model_layers_34_self_attn/std":0.07055711968249352,"train/train/tensor_grad_model_layers_34_mlp_gate_proj_weight/std":4.5556676324713636e-05,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/norm":5.65625,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/max_abs":0.00014972686767578125,"train/train/layer__model_layers_6/param/norm":19.74000934017003,"train/train/tensor_act_model_layers_21_input_layernorm/norm":5792.607421875937,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/max_abs":0.0004119873046875,"train/train/layer_model_layers_91/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_64/act/std":0.6624195279205695,"train/train/tensor_act_model_layers_24_self_attn_k_proj/max_abs":4.21875,"train/train/tensor_act_model_layers_3_mlp_up_proj/max_abs":2.125,"train/train/tensor_act_model_layers_92_mlp_gate_proj/std":0.7539062907658699,"train/train/tensor_param_model_layers_27_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/norm":6.65625,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/mean":-3.619934432208538e-07,"train/train/tensor_act_model_layers_48_input_layernorm/max_abs":6.28125,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/std":6.551663947691921e-05,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/norm":8,"train/train/tensor_param_model_layers_80_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/mean":-8.139340934576467e-08,"train/train/tensor_act_model_layers_66_mlp_down_proj/norm":710.4658632805782,"train/train/tensor_act_model_layers_43_mlp_down_proj/mean":-5.019456148147583e-05,"train/train/tensor_act_model_layers_84_input_layernorm/max_abs":5.75,"train/train/tensor_act_model_layers_45_self_attn_k_proj/mean":0.02117919921875,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/norm":0.0038091237359660583,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/mean":2.2491440176963806e-07,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/std":7.34623405141024e-05,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_70_input_layernorm/mean":0.007293701171875,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/mean":1.9311904907226562e-05,"train/train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_q_proj/norm":4861.567093515534,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/max_abs":0.18359375,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/norm":0.028078825452046235,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/std":6.205917363850628e-05,"train/train/tensor_param_model_layers_0_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_17_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_88_mlp_gate_proj/mean":-0.010986328125,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/norm":0.0006380872292501163,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/mean":-2.6356428861618042e-05,"train/train/tensor_act_model_layers_40_self_attn_q_proj/std":0.9345718981332417,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn_k_proj/std":0.876955327050348,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/std":4.353818364554121e-05,"train/train/tensor_grad_model_layers_54_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_18/grad/max_abs":0.00128936767578125,"train/train/layer_model_layers_85/act/norm":18166.009803804638,"train/train/layer__model_layers_70/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_mlp_gate_proj_weight/mean":7.915496826171875e-05,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/norm":4.375,"train/train/tensor_act_model_layers_72/mean":0.004933834075927734,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/std":0.05224609375,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/mean":1.594889909029007e-08,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/mean":-8.523056749254465e-08,"train/train/tensor_act_model_layers_73_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp/max_abs":0.357421875,"train/train/tensor_act_model_layers_14/std":1.27930237390341,"train/train/tensor_act_model_layers_17_self_attn_q_proj/norm":6210.69175699464,"train/train/tensor_act_model_layers_77_mlp_up_proj/max_abs":2.84375,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/std":0.03271484375,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/norm":0.03717897814390399,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_80_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/mean":4.016910679638386e-07,"train/train/layer_model_layers_50/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_mlp_gate_proj_weight/max_abs":0.1767578125,"train/train/tensor_grad_model_layers_71_mlp_gate_proj_weight/max_abs":0.000579833984375,"train/train/tensor_act_model_layers_74_mlp_down_proj/norm":864.5204021687772,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_post_attention_layernorm/max_abs":5.8125,"train/train/layer__model_layers_65/param/norm":23.996154477071947,"train/train/tensor_act_model_layers_14_self_attn_o_proj/max_abs":0.65625,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/max_abs":0.000911712646484375,"train/train/tensor_param_model_layers_89_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_83_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/max_abs":0.00010585784912109375,"train/train/tensor_act_model/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/max_abs":0.201171875,"train/train/tensor_grad_model_layers_38_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_input_layernorm_weight/mean":1,"train/train/layer_model_layers_11/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/max_abs":0.140625,"train/train/tensor_act_model_layers_46_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/max_abs":0.0003185272216796875,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/norm":4.46875,"train/train/layer_model_layers_88/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_v_proj/mean":-0.0019550323486328125,"train/train/tensor_act_model_layers_74_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_15/param/max_abs":1,"train/train/tensor_grad_model_layers_55_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/norm":0.0226940825715066,"train/train/layer__model_layers_82/param/std":0.060273137041517225,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/std":3.816451704272039e-05,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/mean":4.854518920183182e-07,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/norm":4.21875,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_mlp_gate_proj_weight/max_abs":0.0007476806640625,"train/train/tensor_grad_model_layers_28_mlp_gate_proj_weight/norm":0.015083280186094082,"train/train/tensor_act_model_layers_36_post_attention_layernorm/norm":5792.612182619352,"train/train/tensor_act_model_embed_tokens/mean":-0.0002033710479736328,"train/train/tensor_act_model_layers_2/max_abs":8,"train/train/tensor_act_model_layers_28_self_attn/mean":-0.0007867813110351562,"train/train/tensor_act_model_layers_77_mlp/std":0.17578129506565882,"train/train/tensor_act_model_layers_44_mlp_up_proj/frac_near_user_limit":0,"train/train/global/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn/std":0.13989494957724355,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/mean":2.398155629634857e-07,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/mean":3.0873343348503113e-07,"train/train/tensor_param_model_layers_55_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_self_attn/max_abs":1.125,"train/train/layer_model_layers_85/grad/mean":-1.0661209836114029e-07,"train/train/tensor_act_model_layers_20_post_attention_layernorm/max_abs":6.3125,"train/train/tensor_act_model_layers_75_input_layernorm/mean":0.0081939697265625,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn/norm":711.6674224229876,"train/train/layer_model_layers_62/grad/mean":1.0159843007041587e-07,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/norm":0.010029563863110791,"train/train/tensor_act_model_layers_59_self_attn/mean":0.00028634071350097656,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn/norm":484.1207608406099,"train/train/tensor_param_model_layers_21_mlp_gate_proj_weight/mean":-0.0001163482666015625,"train/train/tensor_grad_model_layers_68_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp_gate_proj/std":0.523437576674253,"train/train/layer_model_layers_48/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_up_proj/std":0.2519533790921045,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/mean":-0.00012874603271484375,"train/train/tensor_grad_model_layers_78_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/std":7.034070190461449e-05,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_q_proj/norm":7133.460083259722,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/mean":-2.0174775272607803e-07,"train/train/layer_model_layers_27/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/std":6.230933354807965e-05,"train/train/tensor_act_model_layers_12_mlp_up_proj/std":0.24145546929146464,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/std":0.0322265625,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/std":0.0311279296875,"train/train/tensor_act_model_layers_90_self_attn_q_proj/mean":-0.0933837890625,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/norm":0.013711938739963873,"train/train/tensor_act_model_layers_37_self_attn_k_proj/mean":-0.04388427734375,"train/train/tensor_act_model_layers_74_self_attn_v_proj/std":0.45752047697181214,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/mean":2.0116567611694336e-05,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_79_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_q_proj/mean":0.115234375,"train/train/tensor_act_model_layers_55_self_attn_k_proj/mean":0.03887939453125,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/norm":6.0625,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/mean":0.000301361083984375,"train/train/tensor_param_model_layers_36_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/norm":0.020597895817093245,"train/train/tensor_grad_model_layers_49_mlp_gate_proj_weight/std":5.265437687685727e-05,"train/train/tensor_param_model_layers_84_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/std":5.224198442632873e-05,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_63_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp_down_proj/mean":0.0088958740234375,"train/train/tensor_param_model_layers_88_mlp_gate_proj_weight/norm":8.6875,"train/train/tensor_act_model_layers_92_self_attn/norm":1470.4452309839792,"train/train/tensor_act_model_layers_89_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69/std":1.4570392559164902,"train/train/tensor_act_model_layers_72_mlp_down_proj/max_abs":1.046875,"train/train/tensor_act_model_layers_90_self_attn_v_proj/norm":2860.2134042027674,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/mean":1.1271913535892963e-07,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/mean":0.000400543212890625,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp/mean":0.0006113052368164062,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_26/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/std":4.811830383078584e-05,"train/train/tensor_act_model_layers_20_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/std":8.029452554681402e-05,"train/train/tensor_act_model_layers_81_self_attn_v_proj/norm":2694.354981249719,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/mean":-3.647804260253906e-05,"train/train/tensor_act_model_layers_70_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_gate_proj_weight/mean":-0.0001087188720703125,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/std":1.776475115782323e-05,"train/train/tensor_act_model_layers_78_self_attn_v_proj/max_abs":2.828125,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/std":3.882063843742462e-05,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/std":0.031005859375,"train/train/layer_model_layers_16/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn/std":0.10278873537730607,"train/train/tensor_act_model_layers_71_input_layernorm/norm":5792.606201174606,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/max_abs":0.208984375,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/norm":3.265625,"train/train/tensor_grad_model_layers_88_mlp_gate_proj_weight/norm":0.03043427350765375,"train/train/layer__model_layers_41/param/std":0.05276992891331779,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/std":0.033203125,"train/train/tensor_act_model_layers_14_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_q_proj/max_abs":5.78125,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/norm":4.96875,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/max_abs":0.000247955322265625,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_input_layernorm/norm":5792.610351567943,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_67_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/norm":7.125,"train/train/layer_model_layers_11/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/max_abs":0.00115966796875,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/max_abs":0.000186920166015625,"train/train/tensor_grad_model_layers_9_mlp_gate_proj_weight/mean":-3.129243850708008e-07,"train/train/tensor_act_model_layers_43_mlp_up_proj/std":0.31689617813399507,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/std":5.0692472023000976e-05,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/std":1.619263547444724e-05,"train/train/tensor_act_model_layers_69_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/max_abs":0.0002841949462890625,"train/train/layer_model_layers_44/act/max_abs":9.375,"train/train/tensor_act_model_layers_23_self_attn_v_proj/max_abs":2.046875,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/mean":-0.00030517578125,"train/train/layer_model_layers_37/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_45/act/norm":13295.938897446456,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/max_abs":0.2421875,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/mean":1.665949821472168e-05,"train/train/tensor_act_model_layers_73_self_attn_v_proj/norm":2316.1382874308106,"train/train/layer_model_layers_43/act/norm":13715.539369925766,"train/train/tensor_act_model_layers_16_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/mean":5.2852556109428406e-08,"train/train/tensor_act_model_layers_5_post_attention_layernorm/std":1.0000001508742455,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/std":0.0286865234375,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_mlp_gate_proj/max_abs":2.765625,"train/train/tensor_act_model_layers_48/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/std":6.653974551294276e-05,"train/train/tensor_act_model_layers_23/mean":-0.024200439453125,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/max_abs":0.169921875,"train/train/tensor_act_model_layers_37_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn_o_proj/max_abs":1.1484375,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/mean":-0.0003185272216796875,"train/train/tensor_act_model_layers_73_self_attn_q_proj/mean":-0.0709228515625,"train/train/tensor_param_model_layers_24_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_92_input_layernorm/norm":5792.61547852059,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/mean":-1.0721851140260696e-07,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_o_proj/max_abs":0.91796875,"train/train/tensor_param_model_layers_88_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_54_self_attn_o_proj/norm":807.951509744037,"train/train/tensor_act_model_layers_38_input_layernorm/std":1.0000005488980919,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/std":0.00012901892144882689,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/mean":3.348104655742645e-07,"train/train/tensor_act_model_layers_64_mlp/mean":0.0005869865417480469,"train/train/tensor_act_model_layers_35/max_abs":9.4375,"train/train/tensor_act_model_layers_53_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/mean":2.6548514142632484e-07,"train/train/tensor_act_model_layers_45_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn_k_proj/mean":0.0088043212890625,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/mean":-5.78584149479866e-08,"train/train/tensor_act_model_layers_69_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_q_proj/max_abs":3.703125,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/mean":-9.655952453613281e-06,"train/train/tensor_act_model_layers_37_mlp/std":0.05322266180375818,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/max_abs":0.00030517578125,"train/train/epoch_time_elapsed":3381.1598489284515,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/norm":0.015321343943079204,"train/train/tensor_act_model_layers_58/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_self_attn_v_proj/max_abs":2.65625,"train/train/tensor_act_model_layers_88_mlp/std":0.3247083954316358,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/mean":9.69739630818367e-08,"train/train/tensor_grad_model_layers_90_mlp_gate_proj_weight/mean":2.996530383825302e-07,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/std":2.5754138659514305e-05,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/mean":0.00018310546875,"train/train/tensor_act_model_layers_91/norm":14772.964209494019,"train/train/tensor_act_model_layers_68_self_attn/mean":0.00295257568359375,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/std":0.00014544553014311876,"train/train/tensor_act_model_layers_92_mlp_up_proj/frac_near_user_limit":0,"train/train/layer__model_layers_80/param/std":0.06166889453445504,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/max_abs":9.918212890625e-05,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/norm":5.3125,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_mlp_gate_proj_weight/mean":-4.7497451305389404e-07,"train/train/tensor_act_model_layers_14_self_attn_k_proj/mean":-0.03668212890625,"train/train/tensor_act_model_layers_18_self_attn/norm":228.74993726103776,"train/train/layer_model_layers_79/act/norm":16214.7322924367,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/std":0.031982421875,"train/train/tensor_act_model_layers_58_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_post_attention_layernorm/std":1.0000001266598622,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/std":0.05224609375,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/std":3.6911452859731356e-05,"train/train/layer_model_layers_76/grad/frac_near_user_limit":0,"eval/samples_per_second":507.343,"train/train/tensor_act_model_layers_54_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_61/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_mlp_gate_proj/mean":-0.005615234375,"train/train/tensor_act_model_layers_47_self_attn_k_proj/mean":0.029815673828125,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/mean":9.834766387939453e-06,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36/max_abs":9.3125,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/mean":9.601935744285583e-07,"train/train/tensor_act_model_layers_88_input_layernorm/norm":5792.610229493992,"train/train/layer__model_layers_8/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_50/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/norm":3.640625,"train/train/layer_model_layers_73/act/norm":15369.930542239184,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/max_abs":0.0004673004150390625,"train/train/tensor_act_model_layers_8_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/max_abs":0.1455078125,"train/train/tensor_param_model_layers_76_mlp_gate_proj_weight/norm":7.3125,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/norm":0.0022571921849403524,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/norm":0.015772038264040857,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_82/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_post_attention_layernorm/std":1.0000012312076858,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/std":4.5681616212904954e-05,"train/train/tensor_act_model_layers_90_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/std":7.695687955424072e-05,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/norm":0.007890741635395337,"train/train/tensor_act_model_layers_44_post_attention_layernorm/max_abs":6.40625,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/norm":0.0022673169140847295,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_16/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37/std":1.2480522757847656,"train/train/tensor_act_model_layers_57_self_attn/mean":0.00200653076171875,"train/train/tensor_param_model_layers_3_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_12_self_attn/mean":-0.0016994476318359375,"train/train/layer__model_layers_3/param/norm":19.26027019053731,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/mean":-0.00012302398681640625,"train/train/tensor_grad_model_layers_83_mlp_gate_proj_weight/norm":0.029087526661721715,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/std":0.00013651527617426881,"train/train/tensor_act_model_layers_84_post_attention_layernorm/mean":0.00749969482421875,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_mlp_down_proj/std":0.10107423570614328,"train/train/tensor_param_model_layers_72_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_72_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_79_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_up_proj/mean":-0.00446319580078125,"train/train/tensor_grad_model_layers_79_mlp_gate_proj_weight/norm":0.02854064209809204,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/norm":0.006365402796513718,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/std":2.534617162452792e-05,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_v_proj/std":0.3027347427219803,"train/train/tensor_act_model_layers_88_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/norm":0.022894774441679526,"train/train/tensor_act_model_layers_82_self_attn/norm":1244.04466746493,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_input_layernorm/norm":5792.607910163649,"train/train/layer_model_layers_90/grad/std":8.667728579274091e-05,"train/train/tensor_act_model_layers_38/mean":-0.0167388916015625,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_post_attention_layernorm/max_abs":5.21875,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/max_abs":0.119140625,"train/train/tensor_act_model_layers_78_input_layernorm/mean":0.00568389892578125,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp_up_proj/frac_near_user_limit":0,"train/train/layer__model_layers_31/param/mean":0.001663609711502718,"train/train/tensor_act_model_layers_92_self_attn_v_proj/mean":-0.002105712890625,"train/train/tensor_act_model_layers_77/std":1.6543036209020134,"train/train/tensor_act_model_layers_31_self_attn_k_proj/mean":0.0107269287109375,"train/train/layer__model_layers_77/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/std":1.4240324379671711e-05,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_mlp_gate_proj_weight/mean":4.211324267089367e-08,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_9_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn/norm":501.76970397443307,"train/train/tensor_act_model_layers_92_input_layernorm/std":1.000001071369633,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_62/act/max_abs":9.75,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/max_abs":0.000431060791015625,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/norm":4.75,"train/train/tensor_act_model_layers_64_mlp_down_proj/std":0.10681206039204338,"train/train/tensor_param_model_layers_70_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp_up_proj/norm":4387.764514922065,"train/train/tensor_act_model_layers_13_mlp/norm":256.28789305245226,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_44_self_attn_q_proj/std":0.910160594221725,"train/train/tensor_act_model_layers_55_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp/norm":330.3591014355224,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_51/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_56_self_attn_k_proj/std":0.9072282065344821,"train/train/tensor_act_model_layers_18_mlp/mean":0.000888824462890625,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/max_abs":0.2314453125,"train/train/layer_model_layers_31/grad/mean":-7.828485435703057e-08,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/max_abs":0.0024871826171875,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/norm":5.40625,"train/train/tensor_act_model_layers_88_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_mlp_gate_proj_weight/max_abs":0.1484375,"train/train/tensor_act_model_layers_59/norm":7612.141830447834,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/max_abs":0.2099609375,"train/train/tensor_act_model_layers_92_mlp_up_proj/std":0.7460940346043229,"train/train/tensor_act_model_layers_87_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn/max_abs":0.96484375,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/mean":-5.977926775813103e-08,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/std":9.15724206377448e-05,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/norm":4.34375,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/std":2.803536633937442e-05,"train/train/tensor_grad_model_layers_56_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_input_layernorm/mean":-0.0104522705078125,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_o_proj/std":0.22632108612640064,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/norm":0.008712113818472086,"train/train/tensor_act_model_layers_71_self_attn_v_proj/max_abs":2.734375,"train/train/tensor_act_model_layers_78_post_attention_layernorm/mean":0.00531768798828125,"train/train/tensor_act_model_layers_60_input_layernorm/max_abs":5.8125,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/std":0.027099609375,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/max_abs":0.000579833984375,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/norm":0.032701005428646804,"train/train/tensor_param_model_layers_35_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/max_abs":0.1337890625,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/mean":-0.00020694732666015625,"train/train/tensor_grad_model_layers_8_mlp_gate_proj_weight/max_abs":0.000934600830078125,"train/train/tensor_param_model_layers_25_mlp_gate_proj_weight/max_abs":0.1259765625,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/std":2.2841816324728843e-05,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/std":0.024658203125,"train/train/layer__model_layers_78/param/max_abs":1,"train/train/tensor_act_model_layers_67_self_attn/norm":706.9723338155231,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_gate_proj/norm":3346.285064431237,"train/train/tensor_grad_model_layers_58_mlp_gate_proj_weight/max_abs":0.000682830810546875,"train/train/tensor_act_model_layers_89/norm":12988.531885926253,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/norm":0.005886037088427107,"train/train/tensor_act_model_layers_93_mlp_gate_proj/mean":-0.0095977783203125,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/mean":4.9591064453125e-05,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/std":6.621460167807153e-05,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/mean":3.159046173095703e-06,"train/train/tensor_act_model_layers_17_self_attn_v_proj/mean":-0.003635406494140625,"train/train/tensor_act_model_layers_36_mlp_gate_proj/norm":2457.183319410019,"train/train/tensor_act_model_layers_72_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/std":0.032958984375,"train/train/tensor_act_model_layers_69_self_attn/std":0.1369670630949227,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/max_abs":0.19140625,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_mlp_gate_proj_weight/std":0.0498046875,"train/train/tensor_act_model_layers_13/max_abs":8.4375,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/max_abs":0.0002574920654296875,"train/train/tensor_act_model_layers_59_post_attention_layernorm/max_abs":5.84375,"train/train/tensor_act_model_layers_87_self_attn_q_proj/max_abs":6.9375,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/std":3.69180244288819e-05,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/std":8.490223519231599e-05,"train/train/tensor_act_model_layers_33_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/mean":0.0003910064697265625,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/norm":0.0024538913018913866,"train/train/tensor_param_model_layers_24_mlp_gate_proj_weight/max_abs":0.12353515625,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/mean":-6.058544386178255e-07,"train/train/tensor_act_model_layers_22_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp_gate_proj/norm":3286.763148814749,"train/train/tensor_act_model_layers_52_mlp_down_proj/std":0.0737304807585093,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/norm":0.04023712296962523,"train/train/tensor_param_model_layers_4_mlp_gate_proj_weight/std":0.0238037109375,"train/train/tensor_act_model_layers_80_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/mean":0.000240325927734375,"train/train/tensor_act_model_layers_12_self_attn_q_proj/norm":5563.453485755455,"train/train/tensor_param_model_layers_43_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_7/param/std":0.0483683228564934,"train/train/tensor_act_model_layers_37_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/std":0.043701171875,"train/train/tensor_act_model_layers_80_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/norm":0.026460341742774684,"train/train/tensor_param_model_layers_65_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_mlp_gate_proj_weight/norm":0.02567027221524208,"train/train/tensor_act_model_layers_61_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn_o_proj/std":0.16528519885606455,"train/train/tensor_act_model_layers_32_mlp/frac_near_dtype_limit":0,"train/train/layer__model_layers_85/param/max_abs":1,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/mean":5.264882929623127e-08,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/norm":5.28125,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/norm":7.90625,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/mean":4.589790478348732e-06,"train/train/tensor_act_model_layers_31_input_layernorm/max_abs":6.5625,"train/train/tensor_param_model_layers_91_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/std":6.807821929457507e-05,"train/train/tensor_act_model_layers_33_mlp_up_proj/std":0.2832033984361759,"train/train/tensor_act_model_layers_58_mlp_gate_proj/norm":3165.537020410197,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/mean":-1.2430245988070965e-07,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/mean":-3.702007234096527e-07,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/norm":0.021431076522129482,"train/train/tensor_act_model_layers_90_self_attn_v_proj/max_abs":3.5,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_mlp_gate_proj_weight/max_abs":0.000591278076171875,"train/train/tensor_act_model_layers_25_mlp/std":0.03759767018355074,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/mean":-1.3050157576799393e-07,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_93_mlp_down_proj/norm":6506.932066345242,"train/train/tensor_act_model_layers_1_mlp_gate_proj/mean":-0.0109710693359375,"train/train/layer__model_layers_57/param/std":0.05541892846045515,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/mean":1.005755621008575e-07,"train/train/tensor_grad_model_layers_16_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp/mean":0.0006818771362304688,"train/train/layer_model_layers_89/grad/frac_near_user_limit":0,"train/train/layer__model_layers_61/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/mean":8.614733815193176e-08,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/max_abs":0.00095367431640625,"train/train/tensor_act_model_layers_80_mlp_down_proj/std":0.19189517685925903,"train/train/tensor_act_model_layers_14_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/max_abs":0.00127410888671875,"train/train/tensor_act_model_layers_87_mlp_up_proj/norm":4930.7644855836525,"train/train/tensor_act_model_layers_59_self_attn_q_proj/max_abs":7.03125,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_9_mlp_gate_proj_weight/max_abs":0.11376953125,"train/train/tensor_grad_model_layers_86_mlp_gate_proj_weight/mean":6.621703505516052e-07,"train/train/tensor_act_model_layers_4_mlp_up_proj/mean":-0.00505828857421875,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_mlp_gate_proj_weight/std":0.0289306640625,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_o_proj/norm":598.3049937312514,"train/train/tensor_param_model_layers_1_mlp_gate_proj_weight/std":0.0235595703125,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_up_proj/mean":0.0047607421875,"train/train/tensor_param_model_layers_13_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/max_abs":0.283203125,"train/train/tensor_act_model_layers_65_self_attn_q_proj/max_abs":6.34375,"train/train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp_gate_proj/norm":4566.274559579478,"train/train/tensor_param_model_layers_87_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_mlp_gate_proj_weight/std":0.0274658203125,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/max_abs":9.393692016601562e-05,"train/train/tensor_act_model_layers_74/std":1.5468783586842205,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/norm":2121.0339854784334,"train/train/layer_model_layers_53/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/mean":-0.00026702880859375,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_up_proj/max_abs":3.140625,"train/train/layer__model_layers_88/param/max_abs":1,"train/train/tensor_act_model_layers_17_mlp_up_proj/norm":2058.949605404876,"train/train/tensor_act_model_layers_30_input_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_68/param/norm":24.316355735389298,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/norm":4.46875,"train/train/tensor_act_model_layers_46_input_layernorm/max_abs":6.34375,"train/train/tensor_act_model_layers_59_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_gate_proj/mean":0.0090789794921875,"train/train/tensor_act_model_layers_70_mlp_down_proj/std":0.1318359485516941,"train/train/tensor_act_model_layers_27_self_attn/norm":376.8328593210019,"train/train/tensor_act_model_layers_88_self_attn_k_proj/mean":0.1005859375,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/mean":1.3475073501467705e-07,"train/train/tensor_act_model_layers_3/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/norm":3.921875,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_48_post_attention_layernorm/std":1.0000003324821038,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/std":0.0361328125,"train/train/tensor_act_model_layers_6_self_attn_v_proj/mean":-0.00391387939453125,"train/train/tensor_act_model_layers_32/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_27_self_attn_k_proj/max_abs":5.40625,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/norm":0.033090287500206365,"train/train/tensor_act_model_layers_30_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_gate_proj/norm":1902.8503239564398,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/mean":-1.574307680130005e-05,"train/train/tensor_act_model_layers_71_self_attn_v_proj/norm":2479.451842300537,"train/train/tensor_act_model_layers_25_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_80/act/std":0.79818644212029,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_29/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_input_layernorm/max_abs":6.03125,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/max_abs":0.255859375,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_93/grad/mean":-1.3627176203542678e-07,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/norm":3.484375,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/std":3.3680549635188616e-05,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/std":3.96662663583905e-05,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/mean":1.2759119272232056e-07,"train/train/layer__model_layers_6/param/std":0.048716601043975204,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/mean":5.402398528531194e-07,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp/mean":-0.0005235671997070312,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/mean":9.506766218692064e-08,"train/train/tensor_grad_model_layers_20_mlp_gate_proj_weight/max_abs":0.0004711151123046875,"train/train/tensor_param_model_layers_55_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_0/grad/mean":8.008502388893163e-08,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_q_proj/max_abs":5.9375,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/std":0.02734375,"train/train/layer_model_layers_88/act/norm":19056.13317622188,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/norm":4.46875,"train/train/tensor_act_model_layers_37_mlp_up_proj/norm":2467.5088309741714,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/max_abs":0.0001964569091796875,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_46/grad/mean":-1.0535205233487622e-07,"train/train/tensor_act_model_layers_76_mlp_gate_proj/max_abs":2.71875,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/max_abs":0.267578125,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/std":2.6122249752946238e-05,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/max_abs":0.000476837158203125,"train/train/layer__model_layers_37/param/mean":0.0017319796795778081,"train/train/tensor_act_model_layers_9_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_q_proj/mean":0.0594482421875,"train/train/tensor_act_model_layers_52_self_attn_o_proj/std":0.04169219450259611,"train/train/layer_model_layers_4/act/mean":-0.002358300345284598,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/norm":0.0436284829965438,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/norm":0.030563746814999862,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/max_abs":0.00034332275390625,"train/train/layer__model_layers_53/param/norm":21.93004235774523,"train/train/tensor_act_model_layers_21_self_attn_o_proj/norm":485.13577003861036,"train/train/tensor_act_model_layers_18/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/std":0.046142578125,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/max_abs":0.2255859375,"train/train/layer_model_layers_51/act/std":0.6292718223997886,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/norm":3.96875,"train/train/tensor_param_model_layers_30_input_layernorm_weight/std":0,"train/train/layer__model_layers_68/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/std":0.053955078125,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_20/param/std":0.04899179094741398,"train/train/tensor_act_model_layers_8_mlp_down_proj/std":0.04138237862212199,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31/std":1.2441464559487283,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/std":6.148162002982402e-05,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/mean":0.000743865966796875,"train/train/tensor_act_model_layers_11_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/max_abs":0.00010395050048828125,"train/train/tensor_act_model_layers_47/norm":7246.099229002189,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_mlp_gate_proj_weight/norm":0.017955750448552264,"train/train/tensor_act_model_layers_5_mlp_gate_proj/max_abs":2.09375,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/max_abs":0.00017452239990234375,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/mean":-2.360902726650238e-07,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/std":0.033447265625,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/mean":-2.9704096959903836e-08,"train/train/tensor_param_model_layers_72_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn/max_abs":2.140625,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/std":6.316032199069044e-05,"train/train/tensor_act_model_layers_85_self_attn_q_proj/max_abs":6.78125,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_input_layernorm/max_abs":6.5,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/norm":6.53125,"train/train/tensor_act_model_layers_45_self_attn/std":0.06750612821338775,"train/train/tensor_grad_model_layers_12_mlp_gate_proj_weight/mean":2.1816231310367584e-07,"train/train/layer__model_layers_25/param/frac_near_user_limit":0,"train/train/layer__model_layers_23/param/mean":0.0014879818825565523,"train/train/tensor_act_model_layers_18_self_attn_q_proj/max_abs":6,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/norm":4.40625,"train/train/tensor_param_model_layers_85_mlp_gate_proj_weight/mean":-1.4424324035644531e-05,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/std":0.06201171875,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/max_abs":0.000461578369140625,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/max_abs":0.000438690185546875,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp/max_abs":0.44140625,"train/train/tensor_act_model_layers_75/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/norm":4.3125,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/mean":-3.1548552215099335e-08,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/mean":-0.0002880096435546875,"train/train/tensor_param_model_layers_11_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/std":1.6610595182409e-05,"train/train/tensor_act_model_layers_92_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_k_proj/std":1.1250000844399102,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_mlp_gate_proj_weight/norm":5.46875,"train/train/tensor_act_model_layers_9/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/std":0.05419921875,"train/train/tensor_act_model_layers_15/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_o_proj/max_abs":2.28125,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/norm":0.027562154733134506,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/norm":0.004873983029515421,"train/train/tensor_param_model_layers_88_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp/norm":308.2524722240277,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/norm":0.001970855137327912,"train/train/tensor_act_model_layers_28/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_mlp_gate_proj/std":0.31835944911727015,"train/train/tensor_act_model_layers_36_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_post_attention_layernorm/mean":-0.0113525390625,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/mean":-4.400499165058136e-08,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/norm":4.375,"train/train/tensor_param_model_layers_12_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_87/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/max_abs":0.0004558563232421875,"train/train/tensor_param_model_layers_65_mlp_gate_proj_weight/std":0.037353515625,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_input_layernorm/max_abs":5.59375,"train/train/layer_model_layers_35/act/std":0.619654177429437,"train/train/layer_model_layers_20/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/mean":4.517845809459686e-06,"train/train/tensor_act_model_layers_18_self_attn_o_proj/norm":228.74993726103776,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/norm":5.28125,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/norm":0.005505082657864334,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/max_abs":0.23046875,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/norm":5.03125,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/std":3.563492571621484e-05,"train/train/tensor_act_model_layers_22/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/max_abs":0.1591796875,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/std":6.64103161683077e-05,"train/train/layer__model_layers_91/param/mean":0.0015720361480474844,"train/train/tensor_act_model_layers_56_self_attn_q_proj/max_abs":7.0625,"train/train/tensor_grad_model_layers_82_mlp_gate_proj_weight/norm":0.028567191750727926,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/max_abs":0.000728607177734375,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/mean":-8.62637534737587e-08,"train/train/tensor_act_model_layers_76_input_layernorm/std":1.0000019520138312,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/norm":0.012292524411559522,"train/train/tensor_act_model_layers_11_mlp_gate_proj/norm":1921.0714478786347,"train/train/tensor_act_model_layers_34_self_attn_v_proj/std":0.34375018093786686,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_17/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn_o_proj/std":0.10131873739386307,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/norm":6.96875,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/norm":0.017190906880586134,"train/train/tensor_act_model_layers_23_input_layernorm/norm":5792.602783205579,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_mlp_gate_proj_weight/max_abs":0.000827789306640625,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/norm":0.007264754181635965,"train/train/tensor_act_model_layers_36_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/norm":0.0010820165376661424,"train/train/tensor_param_model_layers_48_mlp_gate_proj_weight/mean":-6.151199340820312e-05,"train/train/tensor_param_model_layers_1_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_6_self_attn_k_proj/max_abs":4.25,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/mean":-0.00010442733764648438,"train/train/tensor_act_model_layers_30/mean":-0.0196533203125,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_gate_proj/mean":-0.0012359619140625,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/std":0.024169921875,"train/train/tensor_act_model_layers_30_mlp_up_proj/max_abs":1.796875,"train/train/tensor_act_model_layers_18_mlp_gate_proj/std":0.24169993575066848,"train/train/tensor_act_model_layers_7_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85/frac_near_dtype_limit":0,"train/train/layer_model_layers_62/act/norm":14284.226114734505,"train/train/tensor_act_model_layers_21_self_attn_o_proj/std":0.08374116573040855,"train/train/tensor_act_model_layers_37/mean":-0.017059326171875,"train/train/tensor_act_model_layers_39_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/max_abs":0.1943359375,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/mean":5.669891834259033e-06,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_61/act/std":0.6646752023116107,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/mean":-6.228219717741013e-08,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/std":0.0296630859375,"train/train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_16/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/mean":-0.000286102294921875,"train/train/tensor_param_model_layers_2_mlp_gate_proj_weight/max_abs":0.0986328125,"train/train/tensor_act_model_layers_45_mlp_down_proj/norm":350.24356742412095,"train/train/tensor_act_model_layers_30_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/max_abs":0.00011682510375976562,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/std":0.0294189453125,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/norm":5.125,"train/train/tensor_act_model_layers_30_self_attn_o_proj/mean":-0.0002536773681640625,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/std":0.03955078125,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_input_layernorm_weight/std":0,"train/train/layer__model_layers_25/param/max_abs":1,"train/train/tensor_act_model_layers_66_mlp_down_proj/max_abs":0.86328125,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/norm":0.008481814453841283,"train/train/tensor_act_model_layers_68_mlp_down_proj/std":0.13208078087401162,"train/train/tensor_grad_model_layers_62_mlp_gate_proj_weight/norm":0.024997827316286848,"train/train/tensor_act_model_layers_70_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/std":0.045166015625,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65/std":1.3711000174045216,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/norm":4.1875,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/max_abs":0.000438690185546875,"train/train/tensor_act_model_layers_19_mlp/std":0.038391276411898614,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/norm":4.875,"train/train/tensor_act_model_layers_33_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/mean":-5.507469177246094e-05,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp_down_proj/std":0.12268087561286305,"train/train/layer_model_layers_83/act/max_abs":11.25,"train/train/tensor_grad_model_layers_48_mlp_gate_proj_weight/mean":1.6973353922367096e-07,"train/train/tensor_act_model_layers_49_mlp_up_proj/norm":2784.688340651285,"train/train/tensor_act_model_layers_0/std":1.2871140590663535,"train/train/tensor_act_model_layers_60_self_attn_q_proj/norm":4866.487286850195,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/max_abs":0.000736236572265625,"train/train/tensor_act_model_layers_45_mlp_gate_proj/mean":0.00445556640625,"train/train/tensor_act_model_layers_22_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/norm":0.02353259504002599,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp_down_proj/mean":0.0004444122314453125,"train/train/tensor_act_model_layers_31_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_v_proj/mean":0.00022548437118530273,"train/train/tensor_param_model_layers_78_mlp_gate_proj_weight/max_abs":0.1943359375,"train/train/tensor_act_model_layers_25_input_layernorm/std":1.0000017397090561,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/norm":4.6875,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/std":0.0322265625,"train/train/tensor_act_model_layers_81_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_down_proj/norm":221.10912251431404,"train/train/tensor_act_model_layers_51_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_mlp_gate_proj_weight/mean":4.079192876815796e-07,"train/train/tensor_act_model_layers_61_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_gate_proj/mean":0.0035858154296875,"train/train/layer__model_layers_18/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp/mean":0.0004444122314453125,"train/train/tensor_act_model_layers_21_self_attn_q_proj/max_abs":5.8125,"train/train/tensor_grad_model_layers_70_mlp_gate_proj_weight/max_abs":0.00060272216796875,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/std":0.043212890625,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/mean":1.4388933777809143e-07,"train/train/tensor_act_model_layers_81_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_61/grad/norm":0.06275163528752574,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_76_mlp_up_proj/mean":0.0083770751953125,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_91_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_mlp_gate_proj_weight/max_abs":0.000518798828125,"train/train/tensor_act_model_layers_39_self_attn_v_proj/std":0.37744237388297497,"train/train/tensor_act_model_layers_6_mlp_up_proj/mean":-0.006805419921875,"train/train/layer__model_layers_20/param/norm":19.8495685367151,"train/train/tensor_param_model_layers_41_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_mlp_gate_proj_weight/max_abs":0.2197265625,"train/train/tensor_act_model_layers_2_mlp_up_proj/norm":2672.266781869238,"train/train/tensor_act_model_layers_7_mlp_up_proj/norm":1810.6024804427107,"train/train/tensor_act_model_layers_28_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43/std":1.242187928853471,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/std":1.5230048233469448e-05,"train/train/tensor_act_model_layers_23_self_attn_q_proj/std":0.8984419505123119,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/mean":3.2205134630203247e-06,"train/train/tensor_act_model_layers_21_self_attn/max_abs":1.2421875,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_92/act/std":1.0746624469027135,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/norm":0.03669121704247716,"train/train/layer_model_layers_51/grad/max_abs":0.00106048583984375,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/std":0.0252685546875,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/max_abs":0.0003528594970703125,"train/train/layer_model_layers_21/act/std":0.6445469185515916,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/norm":0.0021965105030785927,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/mean":-0.00016689300537109375,"train/train/layer_model_layers_69/act/norm":14883.655491902613,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_13/grad/max_abs":0.0014495849609375,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/norm":0.14041630836388277,"train/train/tensor_act_model_layers_34_mlp/norm":272.3008345377486,"train/train/tensor_act_model_layers_32_input_layernorm/max_abs":6.625,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/std":0.04052734375,"train/train/tensor_act_model_layers_35_mlp/std":0.04925549169468252,"train/train/tensor_act_model_layers_86_mlp/std":0.27002096981450907,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/std":0.055908203125,"train/train/layer__model_layers_86/param/max_abs":1,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn/norm":1462.8881707193352,"train/train/tensor_param_model_layers_41_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/max_abs":0.2578125,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/mean":-5.848705768585205e-07,"train/train/tensor_act_model_layers_33/norm":7183.95277206724,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp/std":0.19189517685925903,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/mean":-3.073364496231079e-08,"train/train/tensor_act_model_layers_48_self_attn/std":0.08068960477653225,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_55/grad/std":5.2450684120454846e-05,"train/train/layer_model_layers_89/act/norm":19286.83838290003,"train/train/tensor_act_model_layers_46_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_gate_proj/std":0.26171894157437353,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/norm":6.59375,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/std":7.32115627924707e-05,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_post_attention_layernorm/norm":5792.615234375895,"train/train/tensor_grad_model_layers_75_mlp_gate_proj_weight/max_abs":0.000499725341796875,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_k_proj/norm":4230.903132487016,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/mean":-2.384185791015625e-05,"train/train/tensor_act_model_layers_8_self_attn/mean":-0.0004100799560546875,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/norm":3.71875,"train/train/tensor_param_model_layers_62_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/norm":0.02928065804826656,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/norm":5.625,"train/train/layer_model_layers_18/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/norm":5.6875,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/max_abs":0.0015716552734375,"train/train/tensor_act_model_layers_58_self_attn/std":0.09631812271311001,"train/train/tensor_act_model_layers_77_self_attn_k_proj/mean":-0.03485107421875,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_61_input_layernorm/max_abs":5.65625,"train/train/tensor_act_model_layers_19_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/std":4.18058781136276e-05,"train/train/tensor_param_model_layers_27_mlp_gate_proj_weight/mean":-8.440017700195312e-05,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/mean":1.3618264347314835e-06,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_39/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/max_abs":0.1923828125,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp_up_proj/std":0.38281256179039685,"train/train/tensor_act_model_layers_20_mlp_gate_proj/norm":1999.2292788710013,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_41/act/mean":0.0009659826755523682,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp/mean":-0.003604888916015625,"train/train/tensor_act_model_layers_64_self_attn/max_abs":1.3984375,"train/train/tensor_act_model_layers_1_self_attn/norm":93.98442972853606,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/mean":9.298324584960938e-05,"train/train/tensor_act_model_layers_30_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_56/param/norm":22.68276465733399,"train/train/tensor_param_model_layers_53_mlp_gate_proj_weight/std":0.032958984375,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/mean":0.00022125244140625,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/norm":7,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/std":0.0235595703125,"train/train/tensor_act_model_layers_86_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_mlp_gate_proj_weight/max_abs":0.140625,"train/train/tensor_param_model_layers_78_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn_o_proj/std":0.09607083543679609,"train/train/tensor_act_model_layers_83_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/max_abs":0.22265625,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/max_abs":0.140625,"train/train/tensor_act_model_layers_84_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/std":0.04638671875,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_58/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/norm":0.012330385651145581,"train/train/tensor_act_model_layers_57_self_attn_q_proj/std":0.931643024177581,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/mean":-8.265487849712372e-09,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/mean":3.218650817871094e-05,"train/train/tensor_act_model_layers_56/std":1.2988332045774424,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/norm":0.007960131626875422,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/std":6.632134246912247e-05,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/max_abs":0.1552734375,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/norm":0.040480437192874204,"train/train/tensor_act_model_layers_93_mlp_up_proj/norm":6875.482409187016,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/norm":0.028169361203344262,"train/train/tensor_act_model_layers_87_mlp/norm":1807.8509727824598,"train/train/tensor_grad_model_layers_79_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/max_abs":0.1884765625,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/max_abs":0.1591796875,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/mean":-8.869171142578125e-05,"train/train/tensor_act_model_layers_80_mlp_up_proj/norm":4126.883643463452,"train/train/tensor_act_model_layers_15_self_attn_o_proj/max_abs":0.8046875,"train/train/layer__model_layers_34/param/std":0.05230428514707492,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_post_attention_layernorm/mean":-0.03216552734375,"train/train/layer_model_layers_38/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn/mean":-0.0007534027099609375,"train/train/tensor_param_model_layers_89_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_59_mlp_up_proj/norm":3256.938489343111,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/std":0.00012478892544292715,"train/train/tensor_grad_model_layers_59_mlp_gate_proj_weight/norm":0.02439047627767202,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/norm":0.017093331117396773,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/max_abs":0.00012969970703125,"train/train/tensor_act_model_layers_78_post_attention_layernorm/norm":5792.609741212556,"train/train/tensor_param_model_layers_60_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_post_attention_layernorm/max_abs":4.875,"train/train/tensor_act_model_layers_40_self_attn_k_proj/norm":4507.57225542182,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/std":0.0322265625,"train/train/tensor_act_model_layers_78_mlp_up_proj/max_abs":2.984375,"train/train/layer_model_layers_80/grad/norm":0.06950141010773112,"train/train/tensor_act_model_layers_9_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_mlp_gate_proj/norm":2900.0218098704886,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/mean":3.809109330177307e-07,"train/train/tensor_act_model_layers_5_self_attn_q_proj/norm":6129.126464295533,"train/train/tensor_grad_model_layers_58_mlp_gate_proj_weight/mean":2.942979335784912e-07,"train/train/tensor_act_model_layers_75_self_attn_k_proj/norm":5312.587171416717,"train/train/tensor_act_model_layers_30_self_attn_k_proj/std":0.7500000521540624,"train/train/tensor_act_model_layers_84_post_attention_layernorm/std":1.000001067848073,"train/train/tensor_act_model_layers_60_mlp_down_proj/max_abs":0.7578125,"train/train/tensor_act_model_layers_51/max_abs":9.6875,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_post_attention_layernorm/max_abs":6.03125,"train/train/tensor_param_model_layers_9_mlp_gate_proj_weight/norm":4.46875,"train/train/tensor_grad_model_layers_52_mlp_gate_proj_weight/norm":0.020332861745034965,"train/train/tensor_act_model_layers_64_self_attn_k_proj/norm":4727.9792354184065,"train/train/tensor_act_model_layers_61_self_attn_k_proj/mean":0.04058837890625,"train/train/tensor_act_model_layers_15_input_layernorm/std":1.0000007427294775,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/max_abs":0.138671875,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_79/param/mean":0.0016768956891088144,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/mean":6.39352947473526e-07,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp/mean":-0.0006780624389648438,"train/train/tensor_grad_model_layers_45_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_19/act/norm":13635.477353890554,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/norm":6.25,"train/train/tensor_act_model_layers_58_self_attn_o_proj/norm":557.4589458849216,"train/train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_55/param/mean":0.0016499339325379656,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_59/act/frac_near_user_limit":0,"train/train/layer__model_layers_5/param/norm":19.634327998113406,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/norm":0.0019052264805593955,"train/train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/norm":5.0625,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/norm":0.015586955723211804,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_o_proj/mean":0.001373291015625,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/std":0.044189453125,"train/train/tensor_act_model_layers_29_self_attn_q_proj/max_abs":6.53125,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_42/grad/norm":0.04115753838709454,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/max_abs":0.000133514404296875,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/mean":-1.3154931366443634e-08,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/std":0.0001174244252921436,"train/train/tensor_act_model_layers_29_post_attention_layernorm/std":1.0000011384247842,"train/train/tensor_act_model_layers_56_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/std":4.228072814340085e-05,"train/train/tensor_param_model_layers_25_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp_down_proj/max_abs":0.51171875,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/norm":5.40625,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/mean":1.878943294286728e-07,"train/train/tensor_act_model_layers_86_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/norm":0.00499424152026192,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_50/grad/norm":0.042166170487560564,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/mean":-0.0003070831298828125,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/norm":0.002022868893976265,"train/train/tensor_act_model_layers_54_mlp_down_proj/max_abs":0.57421875,"train/train/tensor_act_model_layers_87_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_11/param/std":0.049364666500838296,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/max_abs":0.0004596710205078125,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/mean":-5.245208740234375e-05,"train/train/layer_model_layers_56/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/max_abs":0.14453125,"train/train/tensor_param_model_layers_79_input_layernorm_weight/mean":1,"train/train/layer_model_layers_60/grad/std":6.190815735239699e-05,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/max_abs":0.000446319580078125,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/mean":-2.5582266971468925e-07,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_83/grad/mean":-1.226149371559274e-07,"train/train/layer__model_layers_19/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_input_layernorm/std":1.0000012014053996,"train/train/tensor_act_model_layers_76_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/mean":-5.634501576423645e-07,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_40/param/std":0.053030251703008464,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_up_proj/std":0.41015632805369406,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/norm":0.00655474670256112,"train/train/tensor_act_model_layers_23_mlp_gate_proj/std":0.2519531558996928,"train/train/layer_model_layers_24/grad/max_abs":0.0011138916015625,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/max_abs":0.1640625,"train/train/layer_model_layers_64/grad/std":6.829782901843706e-05,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/norm":0.019700934503871927,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/mean":-5.722977221012115e-07,"train/train/tensor_act_model_layers_65_mlp_up_proj/std":0.4238281898234797,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/max_abs":0.000240325927734375,"train/train/tensor_act_model_layers_56_mlp_gate_proj/std":0.3750000322858479,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/std":9.43431537012769e-05,"train/train/tensor_act_model_layers_26_self_attn_k_proj/mean":0.0015110969543457031,"train/train/layer_model_layers_42/act/mean":-0.00016232899257114956,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/norm":3.5625,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/max_abs":0.1123046875,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/std":8.949210026201889e-05,"train/train/tensor_grad_model_layers_73_mlp_gate_proj_weight/mean":2.314336597919464e-07,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/max_abs":0.000335693359375,"train/train/tensor_act_model_layers_74_mlp_down_proj/std":0.1494140780220421,"train/train/tensor_act_model_layers_18_mlp_down_proj/mean":0.000888824462890625,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/mean":-5.806214176118374e-09,"train/train/tensor_act_model_layers_41_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_34/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/norm":0.014687866551309577,"train/train/layer_model_layers_78/act/norm":16545.080550662424,"train/train/tensor_act_model_layers_62_self_attn_v_proj/std":0.373536121766083,"train/train/tensor_act_model_layers_61_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_77/param/norm":24.37656244992308,"train/train/layer_model_layers_5/act/mean":-0.0034958209310259137,"train/train/tensor_act_model_layers_28_mlp_gate_proj/max_abs":1.859375,"train/train/tensor_act_model_layers_53_self_attn_k_proj/max_abs":4.90625,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_22/act/norm":13648.9016805766,"train/train/tensor_act_model_layers_7_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_o_proj/std":0.060791918687008686,"train/train/tensor_act_model_layers_31_self_attn/norm":378.32849028236365,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/norm":5.1875,"train/train/tensor_act_model_layers_29_mlp_up_proj/norm":2208.58984596741,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/norm":0.014609125155592177,"train/train/layer__model_layers_10/param/mean":0.0017944014574547825,"train/train/tensor_act_model_layers_28_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/max_abs":0.0006561279296875,"train/train/tensor_act_model_layers_25_self_attn/mean":0.0005245208740234375,"train/train/tensor_act_model_layers_72_self_attn_o_proj/max_abs":2.078125,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/max_abs":0.0002288818359375,"train/train/tensor_act_model_layers_85_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp_gate_proj/max_abs":1.8671875,"train/train/tensor_param_model_layers_84_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/max_abs":0.00021648406982421875,"train/train/tensor_act_model_layers_81_self_attn/norm":1419.2259170283617,"train/train/tensor_param_model_layers_73_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/max_abs":0.146484375,"train/train/layer_model_layers_8/grad/std":4.674599418827809e-05,"train/train/tensor_act_model_layers_53/max_abs":10,"train/train/tensor_act_model_layers_73_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_v_proj/mean":-4.616379737854004e-05,"train/train/tensor_act_model_layers_16_mlp_gate_proj/std":0.23730475870975656,"train/train/tensor_grad_model_layers_10_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/max_abs":0.146484375,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/norm":0.04045005690637836,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/std":4.991510232686389e-05,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/std":0.0302734375,"train/train/tensor_act_model_layers_69/frac_near_dtype_limit":0,"train/train/layer_model_layers_33/act/mean":-0.005151169640677316,"train/train/tensor_act_model_layers_33_mlp/max_abs":0.380859375,"train/train/tensor_act_model_layers_34_mlp_up_proj/max_abs":1.7265625,"train/train/layer__model_layers_75/param/max_abs":1,"train/train/tensor_act_model_layers_75_mlp_up_proj/mean":-0.012664794921875,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/mean":-1.582084223628044e-07,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/mean":2.459273673593998e-09,"train/train/layer_model_layers_49/grad/max_abs":0.0012664794921875,"train/train/tensor_act_model_layers_53_input_layernorm/mean":-0.005767822265625,"train/train/tensor_act_model_layers_89/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/std":0.05029296875,"train/train/tensor_act_model_layers_84_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/mean":-1.0999792721122503e-07,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/mean":3.218650817871094e-05,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_mlp_gate_proj_weight/max_abs":0.1435546875,"train/train/tensor_act_model_layers_32_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/norm":4.625,"train/train/tensor_act_model_layers_84_input_layernorm/std":1.0000012689262028,"train/train/tensor_act_model_layers_31_post_attention_layernorm/max_abs":6.625,"train/train/tensor_act_model_layers_3_mlp/mean":-0.0050811767578125,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/max_abs":0.255859375,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/mean":-0.0001621246337890625,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/mean":-1.5133991837501526e-08,"train/train/tensor_grad_model_layers_64_mlp_gate_proj_weight/max_abs":0.0006256103515625,"train/train/tensor_param_model_layers_46_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_gate_proj_weight/mean":2.2411346435546875e-05,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/norm":0.00140683221685449,"train/train/tensor_act_model_layers_91_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_o_proj/std":0.039429669432663426,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_37_input_layernorm/std":1.000000663450699,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/norm":5.46875,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/mean":-8.102506399154663e-08,"train/train/tensor_param_model_layers_19_mlp_gate_proj_weight/norm":4.71875,"train/train/tensor_act_model_layers_61_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/norm":2.921875,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/std":3.723363974344195e-05,"train/train/tensor_act_model_layers_45_self_attn/max_abs":1.25,"train/train/layer_model_layers_9/grad/max_abs":0.00164031982421875,"train/train/tensor_grad_model_layers_91_mlp_gate_proj_weight/norm":0.034310662142202276,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_88_self_attn_v_proj/max_abs":2.84375,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/mean":0.000293731689453125,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn/std":0.17260803759536225,"train/train/layer_model_layers_65/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_k_proj/max_abs":4.3125,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/max_abs":0.001251220703125,"train/train/tensor_param_model_layers_90_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/mean":-1.4086253941059113e-07,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/norm":4.625,"train/train/tensor_act_model_layers_83_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_82_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/norm":0.005115769313531402,"train/train/tensor_act_model_layers_19_mlp_up_proj/max_abs":1.640625,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/norm":4.125,"train/train/tensor_act_model_layers_63_mlp/std":0.10668975508587115,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/max_abs":0.1630859375,"train/train/tensor_grad_model_norm_weight/std":0.0008085448606196077,"train/train/layer__model_layers_74/param/norm":23.758283917141405,"train/train/layer__model_layers_71/param/max_abs":1,"train/train/tensor_act_model_layers_44_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/mean":-8.3498889580369e-08,"train/train/tensor_act_model_layers_8/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_gate_proj_weight/norm":4.84375,"train/train/tensor_param_model_layers_70_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_v_proj/max_abs":1.859375,"train/train/tensor_act_model_layers_31_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/max_abs":0.0004558563232421875,"train/train/tensor_act_model_layers_77_self_attn/max_abs":2.625,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_29/grad/mean":-8.332503846208688e-08,"train/train/tensor_act_model_layers_84_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_40_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/mean":0.000274658203125,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/norm":4.125,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/max_abs":0.1982421875,"train/train/layer__model_layers_64/param/max_abs":1,"train/train/tensor_act_model_layers_49_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/max_abs":0.146484375,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/mean":-0.00013637542724609375,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/norm":0.0006430961877882655,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/norm":7.15625,"train/train/tensor_act_model_layers_45_self_attn_v_proj/norm":1861.355190542902,"train/train/layer__model_layers_83/param/std":0.05933651196001379,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/std":0.05126953125,"train/train/layer__model_layers_34/param/norm":21.195126761534123,"train/train/tensor_act_model_layers_4_mlp/norm":387.20331548691627,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/max_abs":0.23828125,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/norm":0.015942360613633688,"train/train/tensor_act_model_layers_93_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_mlp_gate_proj_weight/max_abs":0.00074005126953125,"train/train/tensor_act_model_layers_10_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn/norm":348.2770832677489,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_self_attn_q_proj/mean":0.0615234375,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/max_abs":0.2470703125,"train/train/layer__model_layers_76/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp_gate_proj/mean":0.001552581787109375,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/norm":0.0008400111495954691,"train/train/layer_model_layers_39/grad/mean":-6.543231224307033e-08,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_36/grad/std":5.1714016923733695e-05,"train/train/layer__model_layers_25/param/std":0.0511886285678293,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/mean":-0.000148773193359375,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn_k_proj/std":0.8916032170659535,"train/train/tensor_act_model_layers_80_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/std":0.04150390625,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/mean":5.818437784910202e-07,"train/train/tensor_param_model_layers_38_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/mean":6.031990051269531e-05,"train/train/tensor_act_model_layers_42_mlp_gate_proj/max_abs":2.171875,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_84/grad/std":8.363665981093752e-05,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_17/grad/norm":0.03995432164512385,"train/train/tensor_act_model_layers_44_mlp_down_proj/std":0.061767590520699765,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_mlp_down_proj/mean":0.0058441162109375,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/mean":-5.324545782059431e-07,"train/train/tensor_act_model_layers_71/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/std":0.043212890625,"train/train/tensor_param_model_layers_30_mlp_gate_proj_weight/mean":6.16908073425293e-06,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/norm":0.026816669526721366,"train/train/tensor_act_model_layers_33_input_layernorm/norm":5792.60913086897,"train/train/tensor_act_model_layers_5_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/max_abs":0.255859375,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_93_mlp_up_proj/max_abs":4.75,"train/train/layer__model_layers_62/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34/norm":7187.91535267948,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/std":4.110322210493598e-05,"train/train/layer_model_layers_82/grad/norm":0.06706360868988924,"train/train/layer_model_layers_88/grad/norm":0.06406785272778154,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_11/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp/max_abs":0.447265625,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/mean":-5.882466211915016e-07,"train/train/tensor_act_model_layers_43_mlp_gate_proj/mean":0.002101898193359375,"train/train/tensor_act_model_layers_67_mlp_gate_proj/max_abs":2.515625,"train/train/tensor_act_model_layers_83_self_attn_v_proj/mean":0.0076141357421875,"train/train/tensor_act_model_layers_36_self_attn_k_proj/max_abs":5,"train/train/tensor_act_model_layers_78_self_attn_o_proj/max_abs":2.234375,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_57/act/norm":14420.962333174999,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/max_abs":0.00144195556640625,"train/train/layer_model_layers_69/grad/std":6.515231088337834e-05,"train/train/tensor_act_model_layers_87_mlp_down_proj/norm":1807.8509727824598,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/std":0.048828125,"train/train/layer__model_layers_72/param/std":0.05924963313428616,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19/max_abs":8.4375,"train/train/tensor_act_model_layers_79_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/norm":0.0007450656509012689,"train/train/tensor_act_model_layers_50_post_attention_layernorm/std":1.0000003276799734,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_gate_proj_weight/norm":4.96875,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/norm":8,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/mean":4.604458808898926e-06,"train/train/tensor_act_model_layers_65_mlp_up_proj/norm":3467.3397032172643,"train/train/tensor_param_model_layers_63_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_86_input_layernorm/mean":0.00778961181640625,"train/train/tensor_param_model_layers_64_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_81_mlp_gate_proj_weight/norm":0.029092136879195735,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/std":0.035888671875,"train/train/tensor_act_model_layers_7_mlp_down_proj/max_abs":0.7421875,"train/train/tensor_act_model_layers_50_self_attn_o_proj/mean":0.0007104873657226562,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/norm":0.03687412584206181,"train/train/tensor_act_model_layers_62_self_attn_q_proj/std":0.8789070722788037,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/norm":0.013568933404783804,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/mean":-1.996755599975586e-06,"train/train/tensor_act_model_layers_13_mlp_gate_proj/mean":0.0001825094223022461,"train/train/tensor_act_model_layers_62/norm":7716.623478238858,"train/train/tensor_act_model_layers_64/mean":0.0021286308765411377,"train/train/tensor_grad_model_layers_52_mlp_gate_proj_weight/std":5.622947733491752e-05,"train/train/tensor_param_model_layers_50_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/norm":0.003946462418794276,"train/train/tensor_act_model_layers_89_mlp_up_proj/max_abs":3.703125,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp_down_proj/std":0.03900161763282184,"train/train/tensor_act_model_layers_60_self_attn/max_abs":1.59375,"train/train/tensor_act_model_layers_43_mlp_down_proj/norm":334.58020256501527,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/max_abs":0.000492095947265625,"train/train/layer_model_layers_7/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp_down_proj/mean":-0.00015848875045776367,"train/train/tensor_act_model_layers_54/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/max_abs":0.1298828125,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_76/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/norm":6.21875,"train/train/tensor_act_model_layers_74_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_up_proj/mean":-0.0092315673828125,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76/max_abs":11,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_v_proj/max_abs":2.53125,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/mean":0.00012874603271484375,"train/train/tensor_act_model_layers_50_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/mean":2.6193447411060333e-09,"train/train/tensor_act_model_layers_32_self_attn_q_proj/norm":5323.2613323845435,"train/train/tensor_grad_model_layers_23_mlp_gate_proj_weight/max_abs":0.000736236572265625,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/std":4.3659216014570506e-05,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/mean":0.00011110305786132812,"train/train/tensor_act_model_layers_24/std":1.2597723805446024,"train/train/tensor_act_model_layers_78_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/std":0.03369140625,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_23_self_attn_o_proj/max_abs":1.03125,"train/train/tensor_act_model_layers_68_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_up_proj/norm":2902.1627479346143,"train/train/tensor_act_model_layers_84_self_attn_o_proj/norm":1278.6258776078105,"train/train/tensor_param_model_layers_80_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_self_attn_o_proj/max_abs":1.25,"train/train/tensor_act_model_layers_5_mlp_gate_proj/norm":1859.9686925745607,"train/train/layer_model_layers_78/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_mlp_gate_proj_weight/norm":0.03467613446415766,"train/train/tensor_act_model_layers_44_self_attn_k_proj/norm":4480.28755068398,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_6/param/mean":0.0014776982680869736,"train/train/tensor_act_model_layers_13_mlp_down_proj/norm":256.28789305245226,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/norm":0.0007859873377915608,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/norm":0.0018783268744064605,"train/train/layer_model_layers_11/act/norm":13686.044575938398,"train/train/tensor_act_model_layers_39_input_layernorm/max_abs":6.71875,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/std":5.290645057140322e-05,"train/train/tensor_act_model_layers_28_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/std":4.6095045464942e-05,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/mean":0.0002079010009765625,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_24/param/std":0.04912959141664584,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/std":0.031005859375,"train/train/tensor_act_model_layers_12_self_attn_v_proj/max_abs":2.078125,"train/train/tensor_act_model_layers_66_post_attention_layernorm/mean":0.005718231201171875,"train/train/layer_model_layers_20/grad/max_abs":0.00095367431640625,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/mean":-5.078315734863281e-05,"train/train/layer_model_layers_9/act/std":0.6251032262878922,"train/train/tensor_act_model_layers_43/mean":-0.0164794921875,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/mean":6.952323019504547e-07,"train/train/tensor_act_model_layers_92_mlp_down_proj/frac_near_user_limit":0,"train/train/layer__model_layers_22/param/norm":20.2126210281627,"train/train/tensor_act_model_layers_87_mlp_gate_proj/mean":-0.00788116455078125,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/std":0.048095703125,"train/train/tensor_act_model_layers_7_input_layernorm/mean":-0.03173828125,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/mean":1.2725591659545898e-05,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/max_abs":0.146484375,"train/train/layer_model_layers_77/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_k_proj/norm":5061.762253084566,"train/train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/mean":0.0001811981201171875,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/norm":0.024050949594225424,"train/train/layer_model_layers_43/grad/mean":9.581580316406703e-08,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_58/param/mean":0.0016690044432832,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/mean":-1.791049726307392e-07,"train/train/tensor_act_model_layers_57_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_mlp/norm":925.7319424835587,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_gate_proj_weight/norm":0.02077653214672689,"train/train/tensor_grad_model_layers_72_mlp_gate_proj_weight/mean":2.3283064365386963e-07,"train/train/layer__model_layers_87/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/mean":-3.4831464290618896e-07,"train/train/tensor_act_model_layers_27_mlp_down_proj/mean":0.0005645751953125,"train/train/tensor_param_model_layers_43_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/std":6.000374150418237e-05,"train/train/tensor_act_model_layers_63_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_post_attention_layernorm/max_abs":5.90625,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_34_mlp_gate_proj_weight/norm":0.016486498438309398,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/max_abs":0.00020599365234375,"train/train/layer_model_layers_83/act/norm":17554.509574803644,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/mean":-0.00013446807861328125,"train/train/tensor_param_model_layers_37_mlp_gate_proj_weight/max_abs":0.138671875,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_k_proj/max_abs":5.0625,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/norm":0.022378438285826238,"train/train/layer_model_layers_69/grad/mean":-1.430052351468067e-07,"train/train/tensor_act_model_layers_80_self_attn_q_proj/norm":6958.398246712604,"train/train/tensor_act_model_layers_3/std":1.294926571082321,"train/train/tensor_act_model_layers_14_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/max_abs":0.1650390625,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_85/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp_gate_proj/max_abs":2.09375,"train/train/tensor_act_model_layers_70_mlp_down_proj/norm":765.3281141027179,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/std":0.044921875,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/norm":4.96875,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/std":2.805038216024954e-05,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/max_abs":0.259765625,"train/train/layer_model_layers_82/grad/std":8.279560479021626e-05,"train/train/tensor_act_model_layers_11_self_attn_v_proj/mean":0.001735687255859375,"train/train/tensor_act_model_layers_10_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_mlp_gate_proj/std":0.42724730654953075,"train/train/layer_model_layers_30/act/std":0.6178251152881019,"train/train/tensor_grad_model_layers_38_mlp_gate_proj_weight/max_abs":0.00052642822265625,"train/train/tensor_act_model_layers_34_mlp_gate_proj/max_abs":1.859375,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/std":0.05712890625,"train/train/tensor_act_model_layers_90_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_post_attention_layernorm/std":1.0000003888270994,"train/train/tensor_act_model_layers_63_mlp/max_abs":0.80859375,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/std":4.45597057193248e-05,"train/train/tensor_act_model_layers_57_self_attn_v_proj/std":0.428712089778898,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/max_abs":0.000820159912109375,"train/train/layer_model_layers_41/act/norm":13684.582709910712,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/mean":1,"train/train/layer__model_layers_17/param/std":0.05023775359487092,"train/train/tensor_act_model_layers_70_self_attn_q_proj/mean":-0.0152740478515625,"train/train/tensor_grad_model_layers_89_mlp_gate_proj_weight/std":8.777894779923517e-05,"train/train/tensor_act_model_layers_1_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp_gate_proj/max_abs":2.15625,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/mean":0.000209808349609375,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/norm":7.03125,"train/train/tensor_act_model_layers_33_mlp_gate_proj/max_abs":2.078125,"train/train/tensor_param_model_norm_weight/norm":11.3125,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63/std":1.339850311625278,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/std":0.052490234375,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/mean":-6.984919309616089e-10,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/mean":-9.73232090473175e-08,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/std":0.051513671875,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/std":9.926734413434805e-05,"train/train/tensor_grad_model_layers_2_mlp_gate_proj_weight/std":8.965854826126336e-05,"train/train/tensor_act_model_layers_14_input_layernorm/std":1.0000003590247881,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/max_abs":0.00016689300537109375,"train/train/tensor_act_model_layers_60_mlp_up_proj/std":0.3979501620380947,"train/train/layer_model_layers_30/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/max_abs":0.00018596649169921875,"train/train/tensor_act_model_layers_52_self_attn_k_proj/mean":-0.0015010833740234375,"train/train/tensor_act_model_layers_30_self_attn/max_abs":1.2109375,"train/train/tensor_act_model_layers_39_self_attn_q_proj/std":0.9326188731552788,"train/train/tensor_param_model_layers_37_mlp_gate_proj_weight/norm":5.375,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/max_abs":0.000827789306640625,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/mean":-0.00012159347534179688,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/std":3.8115330945600646e-05,"train/train/layer_model_layers_5/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/std":0.00010003863208913926,"train/train/tensor_act_model_layers_70_input_layernorm/max_abs":5.71875,"train/train/tensor_act_model_layers_85_self_attn_k_proj/std":1.0156251210432714,"train/train/tensor_act_model_layers_23_self_attn_v_proj/std":0.2675781316892073,"train/train/tensor_act_model_layers_3_self_attn_k_proj/std":0.9287125323459398,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/norm":0.019013700245197843,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/std":5.923430462615398e-05,"train/train/tensor_param_model_layers_31_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/max_abs":0.0004749298095703125,"train/train/layer_model_layers_69/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/norm":4.15625,"train/train/tensor_act_model_layers_14_self_attn/mean":0.0008778572082519531,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/norm":3.390625,"train/train/layer_model_layers_12/grad/mean":3.150224034760188e-08,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_o_proj/max_abs":2.03125,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/norm":0.01940447672360783,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/norm":6.59375,"train/train/tensor_act_model_layers_31_mlp_down_proj/norm":256.3926885119202,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_up_proj/max_abs":3.96875,"train/train/tensor_act_model_layers_10_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp_gate_proj/std":0.4863282653192477,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/mean":2.4191103875637054e-07,"train/train/tensor_act_model_layers_70_self_attn_v_proj/norm":2391.4107019089793,"train/train/tensor_act_model_layers_64_self_attn_k_proj/std":0.8154316245415802,"train/train/tensor_param_model_layers_73_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/std":0.03466796875,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_46/param/norm":21.761580968532595,"train/train/tensor_act_model_layers_59_input_layernorm/norm":5792.614501955898,"train/train/tensor_act_model_layers_83_self_attn_q_proj/norm":6592.450770205023,"train/train/tensor_act_model_layers_38_input_layernorm/norm":5792.603149415528,"train/train/tensor_act_model_layers_48_mlp_up_proj/std":0.3339844170824342,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/std":4.407322716801658e-05,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/norm":6.1875,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/norm":9.75,"train/train/tensor_act_model_layers_10_self_attn_q_proj/mean":0.00818634033203125,"train/train/tensor_param_model_layers_33_mlp_gate_proj_weight/max_abs":0.1328125,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_post_attention_layernorm/mean":-0.015655517578125,"train/train/tensor_act_model_layers_79_mlp_down_proj/max_abs":1.2734375,"train/train/tensor_act_model_layers_87_self_attn_k_proj/max_abs":6.6875,"train/train/tensor_act_model_layers_89_self_attn_k_proj/mean":0.05584716796875,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/std":2.5303664545640164e-05,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_q_proj/mean":0.0107421875,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/norm":0.016703238198705905,"train/train/tensor_act_model_layers_9_mlp_down_proj/mean":0.0006113052368164062,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/max_abs":0.00030517578125,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_63/act/mean":0.006271532603672573,"train/train/tensor_act_model_layers_77_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_88_post_attention_layernorm/norm":5792.60766601985,"train/train/tensor_act_model_layers_73_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/max_abs":0.1435546875,"train/train/tensor_act_model_layers_54_mlp_gate_proj/max_abs":2.265625,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/max_abs":0.000518798828125,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/std":6.640943046155275e-05,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/max_abs":0.10400390625,"train/train/tensor_act_model_layers_39_input_layernorm/std":1.0000004894098933,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/std":4.0854057630782185e-05,"train/train/layer_model_layers_65/grad/norm":0.06641340666399467,"train/train/layer__model_layers_88/param/std":0.06142172150784435,"train/train/tensor_act_model_layers_12_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/max_abs":0.0002307891845703125,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/mean":0.00023174285888671875,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/max_abs":0.00017547607421875,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/norm":0.004790220937668817,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/max_abs":0.17578125,"train/train/layer_model_layers_88/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/max_abs":0.158203125,"train/train/tensor_act_model_layers_82_self_attn_k_proj/norm":5523.992555553223,"train/train/tensor_param_model_layers_49_mlp_gate_proj_weight/max_abs":0.1337890625,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/max_abs":0.000774383544921875,"train/train/layer__model_layers_34/param/max_abs":1,"train/train/tensor_act_model_layers_22_self_attn_k_proj/mean":0.021728515625,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/norm":7.28125,"train/train/layer__model_layers_87/param/max_abs":1,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/norm":0.014091952264237012,"train/train/layer_model_layers_36/act/max_abs":9.3125,"train/train/tensor_act_model_layers_19_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/std":0.039306640625,"train/train/layer_model_layers_78/act/std":0.7633204479562411,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/mean":-7.665948942303658e-08,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_v_proj/norm":1793.8065602675972,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/mean":-2.60770320892334e-08,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/mean":1.6604462871327996e-07,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/std":4.823258183611448e-05,"train/train/tensor_act_model_layers_41_mlp_gate_proj/std":0.3144531614199167,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/std":8.03513074319909e-05,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_43_self_attn_v_proj/norm":2255.866267348084,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/mean":-3.841705620288849e-08,"train/train/tensor_act_model_layers_5_self_attn_o_proj/norm":398.87546874903586,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/norm":4.84375,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/mean":5.435943603515625e-05,"train/train/tensor_param_model_layers_22_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/mean":-1.9017606973648071e-06,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/norm":7.40625,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/std":0.0267333984375,"train/train/layer_model_layers_17/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42/norm":7161.494390855458,"train/train/layer__model_layers_91/param/norm":26.64247152691544,"train/train/tensor_act_model_layers_78_input_layernorm/std":1.0000018962582755,"train/train/tensor_grad_model_layers_88_mlp_gate_proj_weight/max_abs":0.00054931640625,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/std":6.236149089368149e-05,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/std":5.384607468649294e-05,"train/train/tensor_act_model_layers_34_self_attn_q_proj/std":0.891603708003493,"train/train/tensor_act_model_layers_27_self_attn_q_proj/mean":-0.005859375,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/max_abs":0.0004558563232421875,"train/train/tensor_grad_model_layers_29_mlp_gate_proj_weight/max_abs":0.000728607177734375,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/mean":2.4780631065368652e-05,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/norm":0.00310871380576295,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/max_abs":0.00087738037109375,"train/train/tensor_act_model_layers_81_input_layernorm/max_abs":6.03125,"train/train/tensor_act_model_layers_35_post_attention_layernorm/std":1.000000765896109,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/norm":7.1875,"train/train/tensor_act_model_layers_43_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_input_layernorm/std":1.0000017623809,"train/train/tensor_act_model_layers_90_self_attn_o_proj/max_abs":3.28125,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/std":3.182169117904717e-05,"train/train/layer_model_layers_93/act/mean":-0.0030730111258370535,"train/train/tensor_act_model_layers_50_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/mean":-1.767277717590332e-05,"train/train/tensor_act_model_layers_61/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/mean":3.952300176024437e-07,"train/train/tensor_act_model_layers_52_mlp/norm":427.2502918834834,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/std":7.59508391190797e-05,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/max_abs":0.259765625,"train/train/tensor_act_model_layers_59_mlp/max_abs":0.6796875,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/max_abs":0.00023555755615234375,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_24_self_attn/max_abs":0.453125,"train/train/tensor_grad_model_layers_58_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp/max_abs":1.7421875,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/std":0.00014404001107436878,"train/train/tensor_act_model_layers_68_mlp_up_proj/std":0.4433594359257237,"train/train/tensor_act_model_layers_11_self_attn_q_proj/norm":5242.75166800528,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/max_abs":0.1484375,"train/train/tensor_act_model_layers_87_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/std":0.00013369524030281923,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/std":2.6392549521402524e-05,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_27_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21/mean":-0.0264892578125,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/max_abs":0.00024127960205078125,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/norm":0.0008784639881615612,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_k_proj/max_abs":6.34375,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/std":0.0245361328125,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/norm":0.0023623173103452595,"train/train/tensor_act_model_layers_59_self_attn_k_proj/max_abs":5.375,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/std":0.0301513671875,"train/train/tensor_act_model_layers_66_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_v_proj/norm":1982.1538055675728,"train/train/tensor_param_model_layers_8_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_43_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn_v_proj/mean":-0.0010251998901367188,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/norm":4.78125,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_93/act/max_abs":26.125,"train/train/layer_model_layers_54/act/norm":14396.332253064213,"train/train/tensor_act_model_layers_39_post_attention_layernorm/max_abs":6.71875,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/mean":8.605420589447021e-06,"train/train/tensor_act_model_layers_39/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp/norm":229.24278567897133,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/max_abs":0.00014019012451171875,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/norm":3.359375,"train/train/tensor_act_model_layers_76_mlp/mean":0.0019989013671875,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/std":0.0289306640625,"train/train/layer_model_layers_65/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/std":0.056396484375,"train/train/tensor_act_model_layers_32_self_attn_v_proj/mean":-0.0013427734375,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/act/mean":-0.013839449201311384,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/norm":0.009182951732343937,"train/train/tensor_param_model_layers_53_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp_down_proj/std":0.10668975508587115,"train/train/tensor_act_model_layers_46_mlp/mean":5.103647708892822e-06,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_input_layernorm/norm":5792.604003909409,"train/train/tensor_param_model_layers_74_mlp_gate_proj_weight/mean":-0.000225067138671875,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/max_abs":0.0002841949462890625,"train/train/tensor_param_model_layers_27_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_54_mlp_down_proj/std":0.08093289445987627,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/max_abs":0.00029754638671875,"train/train/tensor_act_model_layers_35_mlp_gate_proj/norm":2368.7978594974197,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/std":0.05126953125,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/norm":6.375,"train/train/layer_model_layers_14/grad/max_abs":0.00154876708984375,"train/train/tensor_act_model_layers_27_self_attn_q_proj/max_abs":5.9375,"train/train/tensor_act_model_layers_54_mlp_gate_proj/norm":2985.6342158448065,"train/train/tensor_grad_model_layers_31_mlp_gate_proj_weight/mean":-1.1239899322390556e-07,"train/train/tensor_act_model_layers_8_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/max_abs":0.0002689361572265625,"train/train/tensor_act_model_layers_61_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_post_attention_layernorm/std":1.0000019884590312,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/norm":4.8125,"train/train/layer_model_layers_2/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_mlp_gate_proj_weight/mean":-1.3515818864107132e-07,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/norm":0.13398071817907425,"train/train/tensor_act_model_layers_35_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/max_abs":0.1328125,"train/train/tensor_act_model_layers_73_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_post_attention_layernorm/mean":-0.0003581047058105469,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/norm":0.0015274564217938546,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/max_abs":0.000537872314453125,"train/train/tensor_act_model_layers_31_mlp_gate_proj/mean":-0.0014400482177734375,"train/train/tensor_param_model_layers_38_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/mean":5.141191650182009e-08,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/max_abs":0.0002002716064453125,"train/train/tensor_act_model_layers_67_mlp_up_proj/max_abs":2.734375,"train/train/tensor_act_model_layers_12/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_mlp/max_abs":0.5859375,"train/train/tensor_param_model_layers_75_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/mean":-2.3748725652694702e-08,"train/train/layer__model_layers_58/param/norm":22.318539098695506,"train/train/tensor_act_model_layers_11/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_k_proj/max_abs":5.03125,"train/train/tensor_param_model_layers_55_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp/norm":230.98925763157234,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/max_abs":0.1318359375,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_mlp_gate_proj_weight/norm":5.40625,"train/train/tensor_act_model_layers_9_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_input_layernorm/max_abs":5.1875,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_55/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn_k_proj/max_abs":5.375,"train/train/tensor_act_model_layers_64_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/mean":5.990266799926758e-06,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp/std":0.14453125519464036,"train/train/layer_model_layers_76/grad/std":7.802087184825725e-05,"train/train/tensor_act_model_layers_62_self_attn_o_proj/mean":0.00039768218994140625,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/max_abs":0.162109375,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/std":0.02587890625,"train/train/tensor_act_model_layers_85_self_attn_k_proj/max_abs":6.375,"train/train/tensor_act_model_layers_29/norm":7191.621083527659,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/max_abs":0.0001926422119140625,"train/train/tensor_act_model_layers_21/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn_q_proj/mean":-0.00653076171875,"train/train/tensor_act_model_layers_26_mlp_gate_proj/max_abs":1.9296875,"train/train/tensor_act_model_layers_84_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/mean":-0.0018672943115234375,"train/train/tensor_act_model_layers_67_mlp_down_proj/mean":0.00019049644470214844,"train/train/layer_model_layers_3/act/std":0.6439946418827787,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/mean":0.00014495849609375,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/max_abs":0.18359375,"train/train/tensor_act_model_layers_69/frac_near_user_limit":0,"train/train/layer__model_layers_46/param/std":0.0536986384440167,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_66_mlp/std":0.12268087561286305,"train/train/tensor_act_model_layers_9/norm":7469.727011689883,"train/train/tensor_act_model_layers_4_mlp/mean":-0.002819061279296875,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/norm":5.9375,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/max_abs":0.23828125,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/norm":0.008869176885131289,"train/train/tensor_param_model_layers_24_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn_v_proj/norm":2594.63153194078,"train/train/layer_model_layers_20/grad/std":3.815547069062828e-05,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/std":0.0233154296875,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/mean":-3.24249267578125e-05,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/mean":6.845220923423767e-08,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/norm":0.015163858755276871,"train/train/tensor_param_model_layers_49_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp/max_abs":8.4375,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp_down_proj/std":0.06555210286920718,"train/train/layer_model_layers_65/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_self_attn_k_proj/norm":5009.058799062743,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/max_abs":0.00013065338134765625,"train/train/layer_model_layers_60/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_input_layernorm/norm":5792.614135746277,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/max_abs":0.00034332275390625,"train/train/tensor_act_model_layers_77_self_attn/mean":-0.0041961669921875,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/norm":3.65625,"train/train/tensor_act_model_layers_17_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_56/grad/mean":1.2734564901700072e-07,"train/train/tensor_act_model_layers_9_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/norm":0.0051666944739211275,"train/train/layer_model_layers_7/act/mean":-0.007621242531708309,"train/train/tensor_act_model_layers_0_post_attention_layernorm/norm":5792.400634774912,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn_k_proj/norm":4772.6869944210475,"train/train/tensor_param_model_layers_88_mlp_gate_proj_weight/max_abs":0.2099609375,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_mlp_down_proj/max_abs":1.703125,"train/train/tensor_act_model_layers_86_mlp_up_proj/norm":4649.034822581716,"train/train/tensor_act_model_layers_42_mlp_down_proj/std":0.05969251289088695,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/max_abs":0.000652313232421875,"train/train/tensor_grad_model_layers_61_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/mean":-0.002819061279296875,"train/train/tensor_act_model_layers_26_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/mean":-0.0001723766326904297,"train/train/layer_model_layers_51/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_v_proj/max_abs":2.859375,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp_down_proj/norm":799.4549702821303,"train/train/tensor_param_model_layers_24_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn/std":0.13794017358766877,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_up_proj/mean":-0.004322052001953125,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/max_abs":0.000431060791015625,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn/mean":0.0002892017364501953,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/max_abs":0.0004825592041015625,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_21/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_self_attn_k_proj/max_abs":6.25,"train/train/tensor_act_model_layers_23_self_attn/max_abs":1.03125,"train/train/tensor_act_model_layers_46_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/norm":0.007552283657091791,"train/train/tensor_act_model_layers_75_self_attn/std":0.16528519885606455,"train/train/tensor_act_model_layers_38_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/max_abs":0.23828125,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp/std":0.03900161763282184,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/max_abs":0.2138671875,"train/train/tensor_param_model_layers_58_mlp_gate_proj_weight/norm":6.375,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/max_abs":0.1279296875,"train/train/tensor_act_model_layers_45_input_layernorm/norm":5792.434326174123,"train/train/tensor_act_model_layers_9_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_gate_proj_weight/mean":-1.1682510375976562e-05,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/norm":4.90625,"train/train/tensor_act_model_layers_1_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn_o_proj/mean":5.435943603515625e-05,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/std":0.057373046875,"train/train/tensor_act_model_layers_64_mlp_up_proj/std":0.4062500442269584,"train/train/tensor_act_model_layers_1_mlp_down_proj/mean":0.00392913818359375,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/mean":0.00023937225341796875,"train/train/tensor_grad_model_layers_19_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/mean":-0.0008544921875,"train/train/tensor_act_model_layers_53_self_attn_o_proj/mean":0.0024871826171875,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/max_abs":0.0005645751953125,"train/train/tensor_act_model_layers_45_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/norm":4.28125,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/norm":0.01964260780807882,"train/train/tensor_act_model_layers_78/max_abs":11.375,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/max_abs":0.1767578125,"train/train/tensor_act_model_layers_33_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/norm":0.002409596640198824,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/mean":-0.00025177001953125,"train/train/tensor_param_model_layers_61_mlp_gate_proj_weight/mean":-0.00017070770263671875,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_92_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/max_abs":0.00078582763671875,"train/train/tensor_act_model_layers_52_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/mean":-1.0908115655183792e-07,"train/train/tensor_act_model_layers_0_mlp_down_proj/norm":7373.49843335036,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/mean":6.389617919921875e-05,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model/norm":5792.6140136753575,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/mean":-0.0001010894775390625,"train/train/tensor_act_model_layers_41_self_attn_k_proj/mean":0.023529052734375,"train/train/tensor_act_model_layers_40_post_attention_layernorm/std":1.0000004405154808,"train/train/tensor_act_model_layers_34_self_attn_v_proj/mean":0.0006085634231567383,"train/train/layer_model_layers_21/grad/frac_near_user_limit":0,"train/train/layer__model_layers_61/param/norm":23.917460606740843,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_o_proj/mean":0.00015664100646972656,"train/train/tensor_act_model_layers_10_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp/mean":0.000797271728515625,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/norm":0.012329781390794528,"train/train/tensor_act_model_layers_72_self_attn/std":0.154543290292291,"train/train/tensor_act_model_layers_29_self_attn_v_proj/mean":-0.0012197494506835938,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/norm":0.0239934159923616,"train/train/tensor_act_model_layers_15_self_attn/norm":352.022946903137,"train/train/tensor_act_model_layers_79_self_attn/mean":-0.00040531158447265625,"train/train/layer_model_layers_77/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn/norm":584.5732560041911,"train/train/tensor_act_model_layers_78_self_attn_o_proj/mean":-0.0006682872772216797,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/max_abs":6.961822509765625e-05,"train/train/layer_model_layers_71/act/norm":15157.982575429169,"train/train/tensor_act_model_layers_21_self_attn_k_proj/max_abs":4.84375,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/max_abs":0.00041961669921875,"train/train/tensor_act_model_layers_15_mlp/max_abs":0.55859375,"train/train/tensor_act_model_rotary_emb/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_k_proj/mean":0.030029296875,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/max_abs":0.138671875,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_65/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_19_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_mlp_gate_proj_weight/std":7.582909365343332e-05,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/mean":8.519738912582397e-06,"train/train/tensor_act_model_layers_4_mlp_up_proj/std":0.24365285821279364,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_88_mlp/max_abs":2.421875,"train/train/tensor_act_model_layers_90/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn_q_proj/norm":5110.617292891966,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/std":4.907935366667307e-05,"train/train/tensor_act_model_layers_88_self_attn_o_proj/std":0.1928742542415322,"train/train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/norm":4.875,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/max_abs":0.0004291534423828125,"train/train/tensor_act_model_layers_66_post_attention_layernorm/norm":5792.610595708162,"train/train/tensor_act_model_layers_16_mlp_up_proj/norm":1919.6941631908394,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/max_abs":0.1279296875,"train/train/tensor_act_model_layers_54_input_layernorm/std":1.0000003391978105,"train/train/tensor_act_model_layers_47_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp_down_proj/mean":0.001064300537109375,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/mean":8.09086486697197e-08,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn/std":0.02810918224100752,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/max_abs":0.000438690185546875,"train/train/tensor_act_model_layers_14_post_attention_layernorm/std":1.0000008065250243,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/max_abs":0.000583648681640625,"train/train/layer__model_layers_69/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_mlp_gate_proj_weight/mean":-0.000152587890625,"train/train/tensor_act_model_layers_37_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/std":0.031494140625,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/norm":6.1875,"train/train/tensor_param_model_layers_92_mlp_gate_proj_weight/std":0.055908203125,"train/train/tensor_param_model_layers_81_mlp_gate_proj_weight/max_abs":0.1943359375,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/max_abs":0.0004138946533203125,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_v_proj/std":0.35937528982341915,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/mean":-0.000499725341796875,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/max_abs":0.171875,"train/train/tensor_act_model_layers_65_mlp_gate_proj/std":0.42382830161649127,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/max_abs":0.2734375,"train/train/tensor_act_model_layers_39/max_abs":9.4375,"train/train/tensor_act_model_layers_43_mlp/std":0.05767833230516724,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/mean":3.844499588012695e-06,"train/train/tensor_act_model_layers_19_self_attn_q_proj/std":0.9404312099858416,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/mean":-4.488229751586914e-05,"train/train/tensor_act_model_layers_46_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68/max_abs":10.875,"train/train/tensor_act_model_layers_62_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_gate_proj_weight/std":4.354984040022608e-05,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/norm":0.015887118206719075,"train/train/tensor_act_model_layers_42_self_attn_q_proj/norm":5325.876231626008,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/std":8.883586140478444e-05,"train/train/tensor_grad_model_layers_46_mlp_gate_proj_weight/norm":0.018441709587448285,"train/train/layer__model_layers_55/param/norm":21.474554573742036,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp_gate_proj/mean":-0.010406494140625,"train/train/tensor_act_model_layers_12_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84/mean":0.006519317626953125,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/norm":0.017677134256583668,"train/train/tensor_param_model_layers_27_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/max_abs":0.0004444122314453125,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/std":0.06201171875,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/mean":-0.000225067138671875,"train/train/tensor_act_model_layers_30_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/max_abs":0.001556396484375,"train/train/tensor_act_model_layers_27_input_layernorm/std":1.000001537961634,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_47/grad/max_abs":0.0013885498046875,"train/train/tensor_act_model_layers_27_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn_k_proj/max_abs":6.40625,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/mean":5.362380761653185e-09,"train/train/tensor_act_model_layers_22_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_k_proj/mean":0.04376220703125,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/mean":-4.0279701352119446e-08,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/max_abs":0.000629425048828125,"train/train/tensor_act_model_layers_35_input_layernorm/std":1.000000745989104,"train/train/tensor_act_model_layers_44_input_layernorm/norm":5792.609130869485,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/std":0.0380859375,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/norm":0.02045215137516094,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/max_abs":0.00022411346435546875,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/max_abs":0.0003070831298828125,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/mean":2.9907096177339554e-07,"train/train/tensor_act_model_layers_1_mlp_gate_proj/max_abs":3.25,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/std":0.032470703125,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/norm":0.006166476713672296,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/std":0.0322265625,"train/train/tensor_param_model_layers_65_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/norm":0.014475436644976776,"train/train/tensor_act_model_layers_39_self_attn_q_proj/mean":-0.0479736328125,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_12/act/frac_near_user_limit":0,"train/train/layer_model_layers_77/act/norm":17019.949147180738,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/max_abs":0.1669921875,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/norm":6.125,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/std":0.03173828125,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/mean":-0.000347137451171875,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/max_abs":0.1865234375,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/max_abs":0.00116729736328125,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/max_abs":0.0004119873046875,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/norm":0.034676993901160434,"train/train/tensor_act_model_layers_28_self_attn/std":0.0965610683317751,"train/train/tensor_act_model_layers_12_mlp_gate_proj/std":0.23828125641276504,"train/train/tensor_act_model_layers_55_mlp_up_proj/norm":3042.629966171345,"train/train/tensor_act_model_layers_82_post_attention_layernorm/max_abs":5.65625,"train/train/tensor_act_model_layers_82_mlp/norm":1217.726066880047,"train/train/layer_model_layers_68/act/std":0.7036845268031425,"train/train/tensor_act_model_layers_72_self_attn_k_proj/std":0.8847680565503794,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/max_abs":0.000766754150390625,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/mean":-1.1265277862548828e-05,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/norm":5.71875,"train/train/layer__model_layers_54/param/mean":0.0015961621741236837,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/mean":2.641463652253151e-07,"train/train/tensor_act_model_layers_15_self_attn_v_proj/max_abs":1.90625,"train/train/tensor_act_model_layers_52_mlp_up_proj/norm":2892.8774244777674,"train/train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/norm":4.78125,"train/train/tensor_param_model_layers_63_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/norm":0.0264021324017189,"train/train/tensor_act_model_layers_48_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/std":5.3149589295196745e-06,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/norm":0.0014100712525358088,"train/train/tensor_act_model_layers_24_post_attention_layernorm/max_abs":6.3125,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/mean":0.000213623046875,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/norm":0.0007483299511274651,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/max_abs":0.000274658203125,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/norm":6.28125,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/std":3.72244550556082e-05,"train/train/tensor_act_model_layers_39_self_attn_v_proj/norm":2186.011286087031,"train/train/tensor_param_model_layers_4_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_6_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/max_abs":0.00131988525390625,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/mean":4.153698682785034e-06,"train/train/tensor_param_model_layers_40_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp_gate_proj/max_abs":2.5625,"train/train/tensor_act_model_layers_44_mlp_down_proj/max_abs":0.447265625,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_mlp_gate_proj_weight/std":7.285438144498499e-05,"train/train/tensor_grad_model_layers_35_mlp_gate_proj_weight/mean":1.230509951710701e-07,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/std":0.0498046875,"train/train/tensor_act_model_layers_72_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_q_proj/mean":-0.0053863525390625,"train/train/tensor_act_model_layers_66_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_post_attention_layernorm/norm":5792.609497075677,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/std":0.036865234375,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_34_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_69/act/max_abs":10.8125,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_input_layernorm/std":1.0000005117375221,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/max_abs":0.142578125,"train/train/tensor_act_model_layers_31_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/std":0.000213470677118617,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_post_attention_layernorm/mean":0.005859375,"train/train/tensor_act_model_layers_79_mlp_gate_proj/max_abs":2.828125,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/max_abs":0.00067901611328125,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/mean":1.1228257790207863e-07,"train/train/tensor_act_model_layers_28_self_attn_o_proj/mean":-0.0007867813110351562,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_51/act/max_abs":9.6875,"train/train/tensor_act_model_layers_92_post_attention_layernorm/std":1.0000011366785564,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/norm":4.34375,"train/train/tensor_act_model_layers_73_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_50/grad/std":5.2059074120641114e-05,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/std":4.5975206466759054e-05,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/max_abs":0.1826171875,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/std":0.0478515625,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/norm":5.25,"train/train/tensor_act_model_layers_89_self_attn_k_proj/std":0.9394554784511344,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/std":0.0264892578125,"train/train/tensor_param_model_layers_39_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/norm":0.021556559587502414,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/norm":0.015641352917617576,"train/train/tensor_act_model_layers_81_mlp_up_proj/mean":0.0130767822265625,"train/train/tensor_act_model_layers_85_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_gate_proj/max_abs":2.078125,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/norm":0.035724915046670976,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_self_attn_k_proj/max_abs":4.09375,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/mean":0.00024127960205078125,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/std":0.05126953125,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/max_abs":0.2255859375,"train/train/tensor_act_model_layers_76_self_attn_v_proj/norm":2454.84840166544,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_mlp_gate_proj/std":0.3515625573282961,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/norm":0.0006275292949127096,"train/train/tensor_act_model_layers_61_mlp_up_proj/norm":3308.310783463349,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/max_abs":0.001007080078125,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/max_abs":9.775161743164062e-05,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/mean":-6.714253686368465e-08,"train/train/tensor_grad_model_layers_10_mlp_gate_proj_weight/mean":-1.8638093024492264e-07,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/mean":0.00031280517578125,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/mean":-0.00014209747314453125,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/mean":0.0002422332763671875,"train/train/tensor_grad_model_layers_69_mlp_gate_proj_weight/std":7.058037956037292e-05,"train/train/layer_model_layers_52/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_v_proj/mean":-0.00209808349609375,"train/train/tensor_grad_model_layers_81_mlp_gate_proj_weight/std":8.037764831339366e-05,"train/train/tensor_act_model_layers_42_post_attention_layernorm/max_abs":6.5625,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/std":0.038818359375,"train/train/layer__model_layers_84/param/max_abs":1,"train/train/tensor_act_model_layers_74_self_attn_q_proj/std":1.0918021762818146,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/norm":0.015129487556145679,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/std":5.076987070567814e-06,"train/train/tensor_act_model_layers_38_self_attn_v_proj/norm":2278.9586864395646,"train/train/tensor_param_model_layers_42_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/std":6.048862369994201e-05,"train/train/tensor_act_model_layers_80_post_attention_layernorm/std":1.0000016230829276,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/mean":0.000431060791015625,"train/train/tensor_act_model_layers_64_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/norm":193.78448408511994,"train/train/tensor_act_model_layers_52/mean":-0.00846099853515625,"train/train/tensor_act_model_layers_70_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/std":0.0255126953125,"train/train/layer__model_layers_70/param/norm":23.294564145954737,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/std":0.03173828125,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/std":9.105191494299824e-05,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/max_abs":0.00011730194091796875,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/std":0.035400390625,"train/train/tensor_act_model_layers_56_self_attn_k_proj/mean":-0.021636962890625,"train/train/tensor_act_model_layers_45/max_abs":9.4375,"train/train/tensor_act_model_layers_80_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_25/act/norm":13652.29024826637,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/max_abs":0.0001697540283203125,"train/train/tensor_grad_model_layers_25_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp/max_abs":0.404296875,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/max_abs":0.00011444091796875,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/std":0.05078125,"train/train/tensor_act_model_layers_47_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/max_abs":0.15234375,"train/train/layer_model_layers_17/grad/mean":-3.0828084848972266e-08,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/mean":1.3659009709954262e-06,"train/train/tensor_act_model_layers_33_post_attention_layernorm/std":1.0000007845225463,"train/train/tensor_act_model_layers_82_mlp_down_proj/max_abs":1.4765625,"train/train/tensor_act_model_layers_72_self_attn_v_proj/max_abs":2.796875,"train/train/tensor_act_model_layers_79_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp_gate_proj/std":0.5585937706323767,"train/train/layer__model_layers_83/param/norm":24.03834014392425,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_88/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp_gate_proj/norm":3071.2208156920174,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/norm":0.006679498965407919,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/mean":-0.0313720703125,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_1_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn_q_proj/mean":-0.027618408203125,"train/train/tensor_act_model_layers_11_mlp_up_proj/norm":1888.2754588877317,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/mean":0.00018215179443359375,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_23/act/std":0.6133396060308983,"train/train/tensor_act_model_layers_73_mlp_up_proj/max_abs":2.65625,"train/train/tensor_act_model_layers_53_self_attn_q_proj/max_abs":5.71875,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/max_abs":0.0002613067626953125,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/mean":0.0003032684326171875,"train/train/tensor_act_model_layers_45_mlp/mean":0.00061798095703125,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/mean":-1.9883736968040466e-07,"train/train/tensor_act_model_layers_76_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer_model_layers_77/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_51_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_32/act/max_abs":9.125,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/norm":0.006531528178364354,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/norm":4.84375,"train/train/tensor_act_model_layers_35/norm":7164.2781372302115,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/norm":6.8125,"train/train/tensor_act_model_layers_47_mlp_up_proj/std":0.3354503771777189,"train/train/tensor_act_model_layers_13_mlp_gate_proj/max_abs":1.71875,"train/train/layer_model_layers_90/act/std":0.9507419423190602,"train/train/tensor_act_model_layers_72_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn_q_proj/norm":5315.6644754989775,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/norm":5.5625,"train/train/tensor_act_model_layers_34_self_attn_q_proj/norm":5173.148464944478,"train/train/tensor_act_model_layers_66_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29/std":1.240240554719082,"train/train/tensor_act_model_layers_33_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/max_abs":0.00024318695068359375,"train/train/tensor_param_model_layers_91_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_72/max_abs":10.9375,"train/train/tensor_grad_model_layers_39_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp_up_proj/std":0.4750985097044349,"train/train/layer_model_layers_71/grad/std":6.776032019162041e-05,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/max_abs":0.000125885009765625,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/std":4.6270853839493916e-05,"train/train/tensor_act_model_layers_78_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_mlp_gate_proj_weight/std":0.0260009765625,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/std":0.051513671875,"train/train/layer_model_layers_10/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/mean":1.734006218612194e-07,"train/train/tensor_act_model_layers_86_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_o_proj/max_abs":2.015625,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_3/grad/norm":0.05308953268147535,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_input_layernorm/max_abs":6.25,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_mlp_gate_proj_weight/mean":1.8405262380838394e-07,"train/train/tensor_act_model_layers_16_input_layernorm/mean":-0.02874755859375,"train/train/tensor_act_model_layers_81/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/norm":5,"train/train/tensor_act_model_layers_31_self_attn_q_proj/max_abs":5.84375,"train/train/tensor_act_model_layers_57_post_attention_layernorm/norm":5792.611694340551,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/max_abs":0.00022125244140625,"train/train/tensor_grad_model_layers_20_mlp_gate_proj_weight/mean":-7.534981705248356e-08,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/std":0.02783203125,"train/train/tensor_act_model_layers_13_mlp/mean":0.00040435791015625,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_31/grad/max_abs":0.00168609619140625,"train/train/tensor_act_model_layers_5_mlp/max_abs":0.71484375,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/mean":1.9118189811706543e-05,"train/train/layer_model_layers_16/act/std":0.6690365899181023,"train/train/tensor_act_model_layers_24_self_attn_v_proj/mean":5.21540641784668e-06,"train/train/tensor_param_model_layers_14_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/std":0.0001580083502330995,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/norm":0.02447129217609996,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/norm":8.875,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/mean":-1.7762184143066406e-05,"train/train/layer__model_layers_83/param/frac_near_user_limit":0,"train/train/layer__model_layers_49/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_48/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/mean":7.718335837125778e-08,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/max_abs":0.000209808349609375,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/max_abs":0.18359375,"train/train/tensor_act_model_layers_83/mean":0.00981903076171875,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/mean":-4.377216100692749e-08,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/mean":2.4187102098949254e-08,"train/train/tensor_act_model_layers_28_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/max_abs":0.12109375,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_35/act/mean":-0.0022545712334769113,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/max_abs":0.26171875,"train/train/tensor_act_model_layers_5/std":1.294926571082321,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/std":7.916074829871834e-05,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_mlp_gate_proj/norm":2612.6424116711455,"train/train/tensor_act_model_layers_31_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_84/param/std":0.06134248760810589,"train/train/tensor_param_model_layers_92_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_80/param/mean":0.0016388558374366225,"train/train/tensor_grad_model_layers_90_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_mlp_down_proj/std":0.11718755282150012,"train/train/tensor_act_model_layers_62_mlp/max_abs":0.73046875,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/mean":6.771087646484375e-05,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/norm":6.40625,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/mean":1.1444091796875e-05,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_66_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_8/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/std":0.0478515625,"train/train/tensor_act_model_layers_25_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/std":5.30092793478297e-05,"train/train/layer__model_layers_46/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/mean":-1.1653173714876175e-07,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/std":0.00011939039703072086,"train/train/layer_model_layers_76/grad/mean":2.1788934550819084e-07,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/max_abs":0.1328125,"train/train/tensor_param_model_layers_87_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn_k_proj/norm":6158.04793953303,"train/train/tensor_act_model_layers_93_post_attention_layernorm/std":1.000000924802889,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/max_abs":0.0005950927734375,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/max_abs":0.0010223388671875,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/mean":2.110004425048828e-05,"train/train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer__model_layers_74/param/std":0.05863671668133873,"train/train/tensor_act_model_layers_24_self_attn_k_proj/std":0.7695312714818768,"train/train/layer_model_layers_73/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/max_abs":0.000949859619140625,"train/train/tensor_act_model_layers_51_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_mlp_gate_proj/norm":1948.4557340036713,"train/train/layer__model_layers_59/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/norm":0.0045191525652603474,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_embed_tokens_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3/mean":-0.030792236328125,"train/train/tensor_act_model_layers_56_post_attention_layernorm/std":1.0000003727049478,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/mean":0.000171661376953125,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/std":0.048095703125,"train/train/tensor_act_model_layers_54_mlp/mean":0.0005865097045898438,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/norm":0.00117269061683172,"train/train/tensor_act_model_layers_64_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/norm":0.0022838793553464336,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/norm":5792.604858409544,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/std":2.7641668202011634e-05,"train/train/tensor_act_model_layers_66_self_attn_o_proj/norm":1158.0701733151552,"train/train/layer__model_layers_82/param/norm":24.416411137030355,"train/train/tensor_grad_model_layers_27_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_47/act/std":0.6314118792054946,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/max_abs":0.15625,"train/train/tensor_act_model_layers_10_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_64/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/std":2.6154730199853393e-05,"train/train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn/mean":0.0011548995971679688,"train/train/tensor_act_model_layers_28_mlp_down_proj/max_abs":0.357421875,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/max_abs":0.000400543212890625,"train/train/layer_model_layers_5/act/std":0.6661412234211258,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/norm":5.3125,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/std":0.030517578125,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/mean":3.939494490623474e-06,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/std":3.888253708855186e-05,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/norm":0.0006240499622158099,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/mean":-1.6298145055770874e-09,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/mean":0.00023174285888671875,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/norm":0.018110582109034466,"train/train/layer_model_layers_60/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_gate_proj_weight/norm":5.9375,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/std":0.05322265625,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/std":6.54511199618822e-05,"train/train/layer__model_layers_52/param/max_abs":1,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/mean":-0.00034332275390625,"train/train/layer_model_layers_53/grad/std":5.466887961350725e-05,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_gate_proj/norm":3651.749055973153,"train/train/tensor_act_model_layers_91_mlp/std":0.5166044882847162,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/norm":6.9375,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/norm":0.007252886228481278,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/mean":6.504356861114502e-06,"train/train/tensor_act_model_layers_39_input_layernorm/norm":5792.60754395963,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/mean":-0.00023365020751953125,"train/train/tensor_act_model_layers_73_mlp_up_proj/mean":-0.00635528564453125,"train/train/layer_model_layers_22/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54/max_abs":9.9375,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/mean":-2.686283551156521e-08,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/mean":-1.3940036296844482e-05,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/max_abs":0.00136566162109375,"train/train/tensor_act_model_layers_0_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/mean":9.874929673969746e-08,"train/train/layer_model_layers_35/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/norm":5924.521361265128,"train/train/tensor_act_model_layers_35_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_post_attention_layernorm/std":1.0000003260428836,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_60/act/mean":-0.00679046128477369,"train/train/tensor_act_model_layers_68/norm":8357.674419171914,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/mean":0.0002593994140625,"train/train/layer_model_layers_0/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_4/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/mean":-3.62396240234375e-05,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/std":9.559101711451547e-05,"train/train/tensor_grad_model_layers_77_mlp_gate_proj_weight/mean":-3.6670826375484467e-08,"train/train/tensor_act_model_layers_3_self_attn_v_proj/std":0.23437502086162473,"train/train/tensor_param_model_layers_57_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_post_attention_layernorm/mean":-0.0145111083984375,"train/train/tensor_act_model_layers_75_self_attn_v_proj/std":0.4199221280424332,"train/train/layer_model_layers_40/act/max_abs":9.375,"train/train/layer_model_layers_64/act/norm":14353.226378655001,"train/train/tensor_act_model_layers_52/std":1.2734381822307417,"train/train/tensor_act_model_layers_62_self_attn_q_proj/max_abs":5.4375,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_self_attn_k_proj/std":0.9091813346610221,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/norm":0.0008736156465987524,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/norm":0.023594094709695035,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_q_proj/std":1.0859377109746935,"train/train/tensor_act_model_layers_67_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_post_attention_layernorm/mean":-0.0147247314453125,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/std":0.031494140625,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/max_abs":0.2001953125,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/norm":0.01274488015787852,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_93_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_input_layernorm/max_abs":5.8125,"train/train/layer_model_layers_86/grad/std":9.058835702880235e-05,"train/train/tensor_act_model_layers_92_self_attn_o_proj/mean":-0.001979827880859375,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/std":0.0341796875,"train/train/layer_model_layers_89/act/max_abs":12.8125,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/max_abs":0.0013580322265625,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/max_abs":0.0003223419189453125,"train/train/layer_model_layers_14/act/std":0.6428855416678867,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/std":6.154413979917788e-05,"train/train/tensor_param_model_layers_84_mlp_gate_proj_weight/max_abs":0.2041015625,"train/train/layer_model_layers_21/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/max_abs":0.0011444091796875,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/max_abs":0.1826171875,"train/train/tensor_act_model_layers_22_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/max_abs":0.000743865966796875,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/mean":-5.014589987695217e-08,"train/train/tensor_act_model_layers_51_post_attention_layernorm/mean":-0.0066375732421875,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_q_proj/max_abs":6.1875,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/mean":-7.867813110351562e-05,"train/train/tensor_act_model_layers_36_self_attn/std":0.1033943937408615,"train/train/layer__model_layers_4/param/std":0.048114765713301644,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/norm":0.0070978467870571145,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_input_layernorm/max_abs":5.4375,"train/train/layer_model_layers_67/act/frac_near_user_limit":0,"train/train/layer_model_layers_69/grad/norm":0.05279794367185359,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/max_abs":0.00115966796875,"train/train/tensor_act_model_layers_20_post_attention_layernorm/norm":5792.6057128940465,"train/train/tensor_act_model_layers_33_post_attention_layernorm/norm":5792.604980476601,"train/train/tensor_act_model_layers_42_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn/std":0.08386292271840698,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/mean":8.726119995117188e-05,"train/train/tensor_act_model_layers_16_self_attn/norm":485.8187210742259,"train/train/tensor_act_model_layers_67_self_attn/std":0.12219601847799517,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/std":7.761389328616893e-05,"train/train/tensor_act_model_layers_48_mlp_up_proj/mean":0.0010347366333007812,"train/train/layer__model_layers_67/param/mean":0.001597471430595505,"train/train/tensor_act_model_layers_70_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/max_abs":0.0005340576171875,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/max_abs":0.2216796875,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_90_mlp_down_proj/max_abs":3.078125,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_88_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_v_proj/std":0.49023475115028325,"train/train/tensor_act_model_layers_42_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/std":0.0615234375,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/std":8.10719514021077e-05,"train/train/tensor_param_model_layers_92_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/norm":5.53125,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/std":0.03857421875,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/mean":0.00019359588623046875,"train/train/tensor_act_model_layers_23_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/norm":0.009413127061263502,"train/train/tensor_grad_model_layers_89_mlp_gate_proj_weight/norm":0.03181542045513272,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/std":0.032958984375,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/max_abs":0.001007080078125,"train/train/tensor_grad_model_layers_45_mlp_gate_proj_weight/mean":2.3166649043560028e-07,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/std":3.313267721696882e-05,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/std":0.00013699948345672917,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/std":0.0277099609375,"train/train/tensor_param_model_norm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn_o_proj/mean":0.00018143653869628906,"train/train/layer__model_layers_14/param/norm":19.946320785814738,"train/train/tensor_act_model_layers_54_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/max_abs":0.0004596710205078125,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/mean":-9.266659617424011e-07,"train/train/tensor_act_model_layers_25_post_attention_layernorm/max_abs":6.34375,"train/train/tensor_act_model_layers_26_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/std":4.143640334530412e-05,"train/train/tensor_act_model_layers_14_post_attention_layernorm/mean":-0.02862548828125,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44/mean":-0.0155181884765625,"train/train/tensor_act_model_layers_49_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86/mean":0.007732391357421875,"train/train/layer_model_layers_74/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/mean":-9.629875421524048e-07,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/std":0.0235595703125,"train/train/tensor_grad_model_layers_14_mlp_gate_proj_weight/std":3.929370537095501e-05,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_1/act/mean":-0.009778022766113281,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp_up_proj/max_abs":2.28125,"train/train/tensor_act_model_layers_48_mlp_gate_proj/std":0.33593751902713626,"train/train/tensor_act_model_layers_35_input_layernorm/mean":-0.01092529296875,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/std":3.9329288106494014e-05,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_33_mlp_up_proj/mean":-0.004329681396484375,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/norm":0.003945407701042433,"train/train/tensor_act_model_layers_14_self_attn/norm":405.60985724028205,"train/train/tensor_act_model_layers_21_mlp_up_proj/std":0.24707048197856374,"train/train/tensor_act_model_layers_19_self_attn_o_proj/std":0.08364283474223369,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/norm":5.84375,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/max_abs":0.1953125,"train/train/tensor_act_model_layers_87_mlp_gate_proj/norm":4891.625523617922,"train/train/tensor_act_model_layers_72_mlp_up_proj/mean":0.0096588134765625,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/max_abs":0.005950927734375,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/mean":-0.000518798828125,"train/train/tensor_param_model_layers_83_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_42_mlp_gate_proj_weight/mean":-3.341119736433029e-08,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/std":0.027587890625,"train/train/layer_model_layers_79/act/std":0.7482377001862016,"train/train/tensor_act_model_layers_51_input_layernorm/mean":-0.00716400146484375,"train/train/tensor_act_model_layers_42_mlp_gate_proj/std":0.32031253888839395,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/std":5.887881328667997e-05,"train/train/layer_model_layers_58/grad/frac_near_user_limit":0,"train/train/layer_model_layers_48/act/std":0.6281069272401204,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/max_abs":0.0003643035888671875,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/max_abs":0.00147247314453125,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/norm":3.5,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/norm":0.014844660981984965,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/max_abs":0.0001697540283203125,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/norm":0.009946998492237435,"train/train/tensor_act_model_layers_20_self_attn_v_proj/max_abs":1.9296875,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/max_abs":0.1416015625,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/std":0.044921875,"train/train/tensor_act_model_layers_56_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_55_input_layernorm/std":1.0000002504166017,"train/train/tensor_act_model_layers_87_post_attention_layernorm/mean":0.00644683837890625,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/norm":3.3125,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/mean":0.00029754638671875,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/std":0.038818359375,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/norm":0.022819395329616094,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/max_abs":0.00012159347534179688,"train/train/tensor_act_model_layers_82_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_o_proj/norm":1001.2058568754359,"train/train/tensor_act_model_layers_75/max_abs":10.75,"train/train/tensor_param_model_layers_57_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/std":8.672059991936429e-05,"train/train/layer_model_layers_35/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_mlp_gate_proj_weight/std":8.119279222996039e-05,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/max_abs":0.000621795654296875,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/std":0.037109375,"train/train/global/grad/norm":0.6355282850282061,"train/train/tensor_act_model_layers_15/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/mean":9.584426879882812e-05,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/norm":0.005959841581221747,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/mean":-0.0002460479736328125,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/std":3.995190439049683e-05,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/norm":0.0023540855050162425,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/max_abs":0.1064453125,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/std":1.2098060881706137e-05,"train/train/layer__model_layers_63/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/std":8.060539943158116e-05,"train/train/tensor_act_model_layers_34/mean":-0.0161895751953125,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/mean":-0.0117950439453125,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/max_abs":0.1259765625,"train/train/tensor_grad_model_layers_75_mlp_gate_proj_weight/norm":0.02747029585931386,"train/train/layer_model_layers_19/grad/mean":-2.769413353314452e-08,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/max_abs":0.1982421875,"train/train/layer_model_layers_26/grad/std":5.062287468484495e-05,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/std":8.932946445217024e-05,"train/train/tensor_act_model_layers_50_mlp_up_proj/std":0.34960937866285524,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/mean":0.00012969970703125,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/std":7.594433458444015e-05,"train/train/tensor_act_model_layers_60_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/std":1.8007454117418856e-05,"train/train/tensor_act_model_layers_69_self_attn_k_proj/norm":4510.630427569741,"train/train/tensor_act_model_layers_17_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_post_attention_layernorm/norm":5792.606689456731,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/max_abs":0.00072479248046875,"train/train/tensor_act_model_layers_35_mlp_down_proj/norm":285.28552406876935,"train/train/tensor_act_model_layers_11_self_attn/mean":0.0002925395965576172,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/max_abs":0.000514984130859375,"train/train/tensor_act_model_layers_83_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_down_proj/norm":468.6209762802814,"train/train/layer_model_layers_54/act/mean":0.002097538539341518,"train/train/tensor_param_model_layers_60_mlp_gate_proj_weight/norm":6.46875,"train/train/tensor_act_model_layers_3_mlp_gate_proj/norm":2393.159406061193,"train/train/tensor_act_model_layers_32_self_attn_k_proj/std":0.8251971999547466,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/std":5.348829254106154e-05,"train/train/tensor_act_model_layers_39_self_attn_o_proj/norm":601.3627344159112,"train/train/tensor_act_model_layers_25_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp/norm":1196.4930190277144,"train/train/tensor_grad_model_layers_81_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_k_proj/std":0.8125001902763437,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_rotary_emb/std":0.71875,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/max_abs":0.1806640625,"train/train/tensor_act_model_layers_58_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_gate_proj_weight/std":0.0242919921875,"train/train/tensor_act_model_layers_62_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_21/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_90/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/norm":0.018533091776832242,"train/train/layer_model_layers_86/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_gate_proj_weight/std":0.0390625,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/norm":5.90625,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_90/norm":13643.968489720482,"train/train/tensor_act_model_layers_85/norm":11348.231432920555,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/mean":-2.849847078323364e-07,"train/train/tensor_act_model_layers_34/frac_near_user_limit":0,"train/train/layer__model_layers_40/param/norm":21.488777957924924,"train/train/tensor_act_model_layers_55_self_attn_k_proj/max_abs":4.6875,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/mean":-4.738103598356247e-08,"train/train/tensor_act_model_layers_85_mlp_down_proj/std":0.2495122097004952,"train/train/tensor_act_model_layers_58_mlp/std":0.09179691421461662,"train/train/layer__model_layers_27/param/norm":20.54497377135403,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/std":7.97670365723061e-05,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/std":0.03955078125,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/mean":-2.5937333703041077e-07,"train/train/tensor_act_model_layers_26_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_mlp_gate_proj_weight/std":0.025634765625,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/mean":-0.0002803802490234375,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/std":7.036798313367126e-05,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/std":2.4781511835175865e-05,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/max_abs":0.0003070831298828125,"train/train/tensor_act_model_layers_46_self_attn/max_abs":1.5859375,"train/train/tensor_act_model_layers_6/norm":7484.201448476827,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/max_abs":0.1572265625,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/std":1.2049813787468887e-05,"train/train/tensor_act_model_layers_89_self_attn/max_abs":1.984375,"train/train/tensor_param_model_layers_56_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_22_mlp_up_proj/mean":-0.0014095306396484375,"train/train/tensor_act_model_layers_52_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/mean":-0.00010442733764648438,"train/train/tensor_act_model_layers_71_mlp/frac_near_user_limit":0,"_wandb":{"runtime":3382},"train/train/tensor_act_model_layers_70_mlp/mean":-0.0005512237548828125,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_post_attention_layernorm/norm":5792.610107423217,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/max_abs":0.00128936767578125,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/std":4.1785475021137206e-05,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/norm":0.0009663916470697208,"train/train/tensor_act_model_layers_14_mlp_down_proj/max_abs":0.52734375,"train/train/tensor_act_model_layers_44_mlp_down_proj/mean":-0.00022482872009277344,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_44/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_gate_proj/mean":0.0111846923828125,"train/train/layer_model_layers_46/act/max_abs":9.4375,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/mean":3.289896994829178e-07,"train/train/layer_model_layers_71/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_gate_proj/mean":-0.00417327880859375,"train/train/tensor_act_model_layers_35_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_v_proj/max_abs":2.859375,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/mean":2.244114875793457e-05,"train/train/tensor_act_model_layers_77_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/mean":-0.0012607574462890625,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_30/grad/max_abs":0.001617431640625,"train/train/tensor_act_model_layers_88_mlp_up_proj/max_abs":3.515625,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/std":6.121549365312438e-05,"train/train/tensor_act_model_layers_43_self_attn_v_proj/max_abs":3.125,"train/train/tensor_act_model_layers_13_self_attn_k_proj/norm":5494.486923029744,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/mean":1.3350509107112885e-06,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_o_proj/mean":-0.0016994476318359375,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/mean":-1.6205012798309326e-07,"train/train/tensor_act_model_layers_74_self_attn_o_proj/std":0.20703476936681248,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/max_abs":0.177734375,"train/train/layer_model_layers_25/grad/norm":0.037410591870490215,"train/train/layer_model_layers_0/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88/max_abs":12.375,"train/train/layer_model_layers_17/act/std":0.6497106233284389,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/norm":0.013650127552994527,"train/train/tensor_act_model_layers_77_self_attn/std":0.2822284554051737,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_88/act/max_abs":12.375,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/std":0.0267333984375,"train/train/tensor_act_model_layers_20_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25/norm":7311.730890009821,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/max_abs":0.0021514892578125,"train/train/layer_model_layers_47/grad/std":5.3434762915270174e-05,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_41/grad/norm":0.04050916757334302,"train/train/tensor_act_model_layers_24_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/std":0.0255126953125,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/max_abs":0.17578125,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_30_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/mean":1.0046642273664474e-07,"train/train/tensor_param_model_layers_93_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_89_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/mean":2.2761523723602295e-05,"train/train/tensor_act_model_layers_79_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/std":8.132148155737785e-05,"train/train/tensor_act_model_layers_31_mlp_gate_proj/norm":2287.464642081491,"train/train/tensor_act_model_layers_51_mlp_down_proj/std":0.07373049749147254,"train/train/tensor_grad_model_layers_86_mlp_gate_proj_weight/std":8.26229713817933e-05,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/max_abs":0.000293731689453125,"train/train/tensor_act_model_layers_57_mlp/norm":532.8931969174981,"train/train/tensor_act_model_layers_68_input_layernorm/std":1.0000015670363156,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/max_abs":0.1259765625,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/norm":0.021253384432830952,"train/train/tensor_act_model_layers_25_self_attn_k_proj/norm":4707.499209591027,"train/train/tensor_param_model_layers_72_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/mean":2.467632293701172e-05,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp/std":0.23584148330394997,"train/train/layer__model_layers_81/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_input_layernorm_weight/std":0,"train/train/layer__model_layers_63/param/mean":0.0015094060793681748,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_21_mlp/std":0.04119887351964911,"train/train/tensor_act_model_layers_12_mlp_down_proj/std":0.04242028985560395,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64/std":1.3476625129399549,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/norm":0.022828780335217822,"train/train/tensor_act_model_layers_78_post_attention_layernorm/max_abs":5.8125,"train/train/tensor_act_model_layers_56_input_layernorm/mean":-0.0012311935424804688,"train/train/tensor_param_model_layers_8_mlp_gate_proj_weight/norm":4.40625,"train/train/tensor_act_model_layers_6_mlp/max_abs":1.2265625,"train/train/tensor_act_model_layers_66_self_attn_o_proj/max_abs":4,"train/train/layer_model_layers_67/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_post_attention_layernorm/max_abs":5.21875,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/std":0.9238306009207004,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/max_abs":0.0002231597900390625,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn_q_proj/max_abs":5.375,"train/train/tensor_act_model_layers_6_mlp_gate_proj/mean":-0.0055999755859375,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/mean":0.00018405914306640625,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/std":4.424540343736316e-05,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/norm":0.00041562041420956726,"train/train/layer_model_layers_39/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/std":1.5416209219468643e-05,"train/train/tensor_act_model_layers_4_self_attn_k_proj/mean":0.0836181640625,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_64/grad/norm":0.05534325468864614,"train/train/tensor_act_model_layers_38_self_attn_o_proj/norm":587.5135877209094,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/norm":0.0010057172802212901,"train/train/layer__model_layers_37/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_84/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/std":8.308189091922124e-06,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/max_abs":0.000438690185546875,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/max_abs":0.0001583099365234375,"train/train/tensor_act_model_layers_29_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_v_proj/max_abs":2.84375,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13/mean":-0.03338623046875,"train/train/tensor_grad_model_layers_20_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_q_proj/max_abs":7.75,"train/train/tensor_act_model_layers_8_mlp_up_proj/norm":1896.7066438807458,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/mean":6.6570937633514404e-06,"train/train/layer_model_layers_59/grad/std":6.962122869720949e-05,"train/train/tensor_act_model_layers_7_self_attn_v_proj/norm":1594.1250802037496,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/std":0.06396484375,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp_up_proj/std":0.27880995776398965,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/max_abs":0.0006866455078125,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/mean":4.553794860839844e-05,"train/train/tensor_act_model_layers_79_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn_v_proj/std":0.30859377697178947,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/max_abs":0.00037384033203125,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/mean":2.2223684936761856e-07,"train/train/tensor_act_model_layers_90_input_layernorm/std":1.0000009094360807,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/norm":4.78125,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/std":8.465321348011578e-05,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_17/act/max_abs":8.375,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_mlp_gate_proj_weight/std":7.893305875053751e-05,"train/train/tensor_act_model_layers_18_self_attn_v_proj/std":0.2910157133198329,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/mean":-0.0003604888916015625,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/std":5.5739323023268645e-05,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/mean":7.507391273975372e-06,"train/train/tensor_act_model_layers_84_mlp_gate_proj/max_abs":3.21875,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_mlp_up_proj/mean":-0.0007114410400390625,"train/train/tensor_act_model_layers_60/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/max_abs":0.000514984130859375,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/std":0.0001303747979797638,"train/train/tensor_act_model_layers_73/mean":0.007465362548828125,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/max_abs":0.1748046875,"train/train/tensor_act_model_layers_17_post_attention_layernorm/max_abs":6.15625,"train/train/tensor_param_model_layers_12_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_8/param/max_abs":1,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/max_abs":0.00054168701171875,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/mean":6.32135197520256e-08,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/max_abs":0.000255584716796875,"train/train/tensor_act_model_layers_50_self_attn/mean":0.0007104873657226562,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/norm":0.01697808919502975,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/norm":5.78125,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/norm":0.014488073572237282,"train/train/tensor_param_model_layers_62_mlp_gate_proj_weight/mean":-1.8596649169921875e-05,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/mean":-6.505433702841401e-08,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/norm":0.010068257328556008,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/mean":-9.257346391677856e-06,"train/train/layer__model_layers_52/param/norm":21.47623709511573,"train/train/tensor_param_model_layers_64_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/max_abs":0.1064453125,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/std":0.03466796875,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/norm":5.90625,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/mean":-1.1621159501373768e-07,"train/train/tensor_param_model_layers_72_mlp_gate_proj_weight/std":0.039306640625,"train/train/layer__model_layers_53/param/max_abs":1,"train/train/layer__model_layers_1/param/mean":0.0016072432448079174,"train/train/tensor_act_model_layers_82_self_attn_v_proj/max_abs":2.734375,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/mean":-1.5401747077703476e-07,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/mean":-1.3008713722229004e-05,"train/train/layer_model_layers_67/act/mean":0.0007623604365757533,"train/train/tensor_act_model_layers_67_mlp_down_proj/std":0.12060549789506718,"train/train/layer__model_layers_72/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/max_abs":0.1376953125,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/max_abs":0.000324249267578125,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/std":2.172101761037662e-05,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_post_attention_layernorm/max_abs":5.6875,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/std":7.308105801011855e-05,"train/train/tensor_act_model_layers_65_mlp_down_proj/max_abs":1.0546875,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/norm":4.75,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_q_proj/max_abs":6.09375,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/std":6.977222982418274e-05,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_24/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/norm":7.4375,"train/train/tensor_act_model_layers_31_self_attn/std":0.06531000439404724,"train/train/tensor_act_model_layers_68_mlp_up_proj/mean":-0.000621795654296875,"train/train/tensor_act_model_layers_60_self_attn_o_proj/mean":-0.0005841255187988281,"train/train/tensor_param_model_layers_26_mlp_gate_proj_weight/mean":-0.000209808349609375,"train/train/tensor_act_model_layers_67_mlp_up_proj/mean":0.0009174346923828125,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/mean":0.0003719329833984375,"train/train/tensor_act_model_layers_5_self_attn/max_abs":0.80859375,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/norm":0.0316484794757671,"train/train/tensor_act_model_layers_62_self_attn_o_proj/std":0.09497983871107281,"train/train/tensor_act_model_layers_30_self_attn_q_proj/norm":5242.193813360873,"train/train/tensor_act_model_layers_57_input_layernorm/std":1.000000394516896,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/norm":0.004196527638976227,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/max_abs":0.2373046875,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/mean":8.20159912109375e-05,"train/train/tensor_act_model_layers_15_self_attn_v_proj/std":0.286622354431158,"train/train/tensor_act_model_layers_63_post_attention_layernorm/std":1.0000005689170537,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/max_abs":0.142578125,"train/train/tensor_act_model_layers_78_input_layernorm/max_abs":5.8125,"train/train/tensor_act_model_layers_30_self_attn_o_proj/max_abs":1.2109375,"train/train/tensor_act_model_layers_7_self_attn_q_proj/mean":-0.0316162109375,"train/train/tensor_act_model_layers_1_mlp/mean":0.00392913818359375,"train/train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/max_abs":0.000331878662109375,"train/train/tensor_param_model_layers_19_mlp_gate_proj_weight/max_abs":0.1259765625,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/max_abs":0.0003452301025390625,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51/std":1.2656257910490507,"train/train/tensor_act_model_layers_1/std":1.279301602231093,"train/train/layer_model_layers_89/act/std":0.8898258655940564,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/std":3.472452677734358e-05,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/std":6.892053548732359e-05,"train/train/layer__model_layers_72/param/mean":0.001709960366188085,"train/train/layer_model_layers_59/grad/norm":0.05640191381295426,"train/train/tensor_param_model_layers_33_mlp_gate_proj_weight/norm":5.25,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/norm":0.00130679413367008,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/max_abs":0.1015625,"train/train/tensor_act_model_layers_91_mlp/mean":0.0019741058349609375,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/std":0.00011493493005759506,"train/train/tensor_act_model_layers_35_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/std":0.0419921875,"train/train/tensor_act_model_rotary_emb/max_abs":1,"train/train/tensor_act_model_layers_19/norm":7360.5710055084255,"train/train/tensor_act_model_layers_89_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/max_abs":0.2421875,"train/train/tensor_param_model_layers_57_mlp_gate_proj_weight/norm":6.3125,"train/train/tensor_act_model_layers_78_mlp_gate_proj/max_abs":2.859375,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_norm_weight/std":0,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/mean":1.635635271668434e-08,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn/std":0.07020243010870277,"train/train/tensor_act_model_layers_45_self_attn_o_proj/norm":390.63338948877305,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/mean":0.0003185272216796875,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_k_proj/norm":6215.001225485138,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/mean":-4.351139068603516e-06,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/std":0.028076171875,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/mean":-0.000274658203125,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_39/grad/std":5.273640652613429e-05,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/norm":6.96875,"train/train/tensor_act_model_layers_91_mlp/max_abs":4.09375,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/std":4.148130657108905e-05,"train/train/tensor_act_model_layers_23_mlp_up_proj/std":0.2519531490844338,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/norm":0.031392309881310414,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/max_abs":0.0003814697265625,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/max_abs":0.000865936279296875,"train/train/tensor_act_model_layers_61_self_attn_k_proj/std":0.7959032561735541,"train/train/tensor_act_model_layers_58_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/max_abs":0.0002880096435546875,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/max_abs":0.00052642822265625,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/std":0.00013208721428203742,"train/train/tensor_act_model_layers_91_self_attn_q_proj/max_abs":6.5625,"train/train/tensor_act_model_layers_29_post_attention_layernorm/norm":5792.610107425975,"train/train/layer_model_layers_56/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/std":6.340260885101584e-05,"train/train/tensor_act_model_layers_9_input_layernorm/mean":-0.03192138671875,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/estimated_remaining_minutes":0,"train/train/tensor_act_model_layers_71_mlp/std":0.1379401859964465,"train/train/tensor_act_model_layers_39_post_attention_layernorm/norm":5792.607177735306,"train/train/tensor_act_model_layers_64_input_layernorm/max_abs":5.53125,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/norm":4.3125,"train/train/global/act/norm":178842.9867831975,"train/train/tensor_grad_model_layers_54_mlp_gate_proj_weight/mean":1.9755680114030838e-07,"train/train/tensor_act_model_layers_12_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/max_abs":0.0002880096435546875,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_post_attention_layernorm/max_abs":5.75,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/std":0.0306396484375,"train/train/tensor_grad_model_layers_62_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10/max_abs":8.8125,"train/train/tensor_act_model_layers_0_mlp_up_proj/std":0.7587910520804892,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/norm":4.46875,"train/train/tensor_act_model_layers_4_self_attn_o_proj/max_abs":1.078125,"train/train/tensor_act_model_layers_51_post_attention_layernorm/norm":5792.611938480435,"train/train/tensor_act_model_layers_60_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp_gate_proj/norm":3699.322984083319,"train/train/tensor_act_model_layers_58_mlp_down_proj/max_abs":0.6640625,"train/train/tensor_act_model_layers_41_self_attn/max_abs":1.5546875,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_post_attention_layernorm/mean":-0.02496337890625,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp/mean":-0.00180816650390625,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/mean":0.00013065338134765625,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/norm":0.017503048256649823,"train/train/layer__model_layers_36/param/std":0.05348013324106412,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/std":0.05810546875,"train/train/tensor_act_model_layers_41_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer__model_layers_92/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_post_attention_layernorm/mean":-0.0123443603515625,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/norm":0.0010082642500127178,"train/train/tensor_act_model_layers_26_mlp_gate_proj/std":0.2539063238825255,"train/train/layer_model_layers_90/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/max_abs":0.123046875,"train/train/tensor_act_model_layers_80_mlp_up_proj/std":0.5039065697828607,"train/train/tensor_param_model_layers_31_mlp_gate_proj_weight/mean":2.658367156982422e-05,"train/train/tensor_act_model_layers_41_mlp_up_proj/std":0.31640632291909987,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/norm":0.0006603669582937077,"train/train/tensor_act_model_layers_13_self_attn/max_abs":0.6953125,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/mean":8.106231689453125e-05,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/mean":-1.811981201171875e-05,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/max_abs":0.000278472900390625,"train/train/layer_model_layers_79/grad/mean":1.4568169785782624e-07,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/mean":-6.370246410369873e-07,"train/train/tensor_act_model_layers_56_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/norm":0.017432243272638027,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/std":0.0299072265625,"train/train/layer_model_layers_13/act/std":0.6517599219007854,"train/train/tensor_act_model_layers_5_mlp_up_proj/std":0.22876018800971284,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/std":9.049235435218757e-05,"train/train/tensor_act_model_layers_66_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_q_proj/mean":0.036376953125,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/mean":-3.0925730243325233e-07,"train/train/tensor_param_model_layers_78_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/mean":1.623295247554779e-06,"train/train/tensor_act_model_layers_22_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn/max_abs":0.53125,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_2_post_attention_layernorm/norm":5792.608032229273,"train/train/layer_model_layers_27/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn_k_proj/mean":-0.02276611328125,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/max_abs":0.0002384185791015625,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/max_abs":0.000370025634765625,"train/train/tensor_act_model_layers_9_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_k_proj/norm":4698.308027305562,"train/train/tensor_act_model_layers_88_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39/mean":-0.017822265625,"train/train/tensor_act_model_layers_92_self_attn_q_proj/max_abs":6.28125,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/norm":0.006865692096281639,"train/train/tensor_act_model_layers_71_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/norm":0.007663224556520989,"train/train/layer__model_layers_39/param/norm":21.127022369911714,"train/train/tensor_act_model_layers_26_mlp/max_abs":0.36328125,"train/train/tensor_param_model_layers_31_input_layernorm_weight/mean":1,"train/train/layer__model_layers_12/param/std":0.049243907898492575,"train/train/tensor_act_model_layers_14_post_attention_layernorm/norm":5792.612915039694,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_8/mean":-0.0328369140625,"train/train/layer_model_layers_44/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_up_proj/norm":6217.584576125047,"train/train/tensor_act_model_layers_35_self_attn_k_proj/std":0.813478698246821,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_up_proj/frac_near_user_limit":0,"train/train/layer__model_layers_53/param/mean":0.001639450954014723,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/std":0.040283203125,"train/train/tensor_act_model_layers_46_self_attn_v_proj/std":0.3754892690271052,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/norm":0.03723404672008132,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/max_abs":0.21875,"train/train/tensor_act_model_layers_60_mlp/norm":580.2673105401778,"train/train/tensor_act_model_layers_87_self_attn_v_proj/mean":0.00336456298828125,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/mean":0.0001163482666015625,"train/train/tensor_act_model_layers_88_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn/mean":-0.0003643035888671875,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/max_abs":0.00069427490234375,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/mean":0.0001239776611328125,"train/train/tensor_act_model_layers_17_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn/norm":644.836249174331,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_mlp_gate_proj_weight/std":0.035888671875,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/std":2.217080201579336e-05,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/max_abs":0.1298828125,"train/train/tensor_act_model_layers_8_mlp_gate_proj/mean":-0.0001308917999267578,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp_gate_proj/std":0.33007813194358837,"train/train/tensor_act_model_layers_43_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/norm":3.15625,"train/train/tensor_act_model_layers_39_mlp_gate_proj/max_abs":1.8515625,"train/train/tensor_act_model_layers_84_self_attn_v_proj/norm":2730.6203337633815,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/std":0.00013056635523675612,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_input_layernorm/norm":5792.603271486705,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/norm":4.3125,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/norm":6.0625,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/mean":0.000278472900390625,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/max_abs":8.916854858398438e-05,"train/train/layer__model_layers_8/param/std":0.04830752644402853,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/std":4.226612041907585e-05,"train/train/tensor_act_model_layers_30_self_attn_v_proj/mean":0.0007495880126953125,"train/train/tensor_act_model_layers_87_post_attention_layernorm/norm":5792.611572269581,"train/train/tensor_act_model_layers_21_mlp/max_abs":0.5234375,"train/train/tensor_grad_model_layers_47_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_gate_proj/std":0.3642593730836835,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_o_proj/norm":348.2770832677489,"train/train/tensor_act_model_layers_13_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_gate_proj/std":0.40820321330184756,"train/train/tensor_act_model_layers_23_input_layernorm/max_abs":6.28125,"train/train/tensor_act_model_layers_74_self_attn_k_proj/norm":5727.499267276459,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_40/grad/max_abs":0.001312255859375,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/mean":2.0570587366819382e-07,"train/train/layer_model_layers_12/grad/norm":0.039049650245029655,"train/train/tensor_act_model_layers_79_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/std":2.5199942033227118e-05,"train/train/tensor_act_model_layers_48_mlp/norm":379.68714743653817,"train/train/tensor_param_model_layers_11_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_43_self_attn_o_proj/norm":644.836249174331,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/mean":2.859160304069519e-07,"train/train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_q_proj/norm":6478.116989747306,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/norm":7.4375,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_10_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_26/act/std":0.6353081045925545,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/max_abs":0.248046875,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/norm":3.140625,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/norm":4.21875,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/max_abs":0.00015735626220703125,"train/train/layer_model_layers_45/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/std":0.039794921875,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/norm":0.023663149821431462,"train/train/tensor_act_model_layers_50_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/std":3.650127355799336e-05,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/max_abs":0.0004062652587890625,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/max_abs":0.2138671875,"train/train/tensor_act_model_layers_57_self_attn_k_proj/norm":5173.425855991581,"train/train/tensor_act_model_layers_39_self_attn_o_proj/std":0.10388278092247015,"train/train/tensor_act_model_layers_17_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/mean":-0.0003871917724609375,"train/train/tensor_grad_model_layers_85_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_k_proj/std":0.8183618035655984,"train/train/layer__model_layers_80/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/norm":6.25,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/max_abs":0.166015625,"train/train/tensor_param_model_layers_44_mlp_gate_proj_weight/norm":5.65625,"train/train/layer_model_layers_57/act/std":0.6653286552555148,"train/train/tensor_act_model_layers_56_mlp_down_proj/norm":491.13860279983805,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/norm":0.006629763865250218,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/max_abs":0.000789642333984375,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_mlp_gate_proj_weight/std":4.460285151250836e-05,"train/train/tensor_act_model_layers_22_post_attention_layernorm/mean":-0.0205078125,"train/train/layer_model_layers_28/act/std":0.6463911803349374,"train/train/tensor_act_model_layers_91_self_attn_k_proj/mean":0.04254150390625,"train/train/tensor_act_model_layers_37_mlp_down_proj/mean":0.000457763671875,"train/train/layer__model_layers_26/param/max_abs":1,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/mean":0.0005035400390625,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/norm":0.012031436646239875,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/norm":0.025943746848624268,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/mean":0.0003261566162109375,"train/train/tensor_param_model_layers_35_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_87_mlp_gate_proj_weight/std":0.0478515625,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/std":4.001525694501789e-05,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/std":0.056884765625,"train/train/tensor_act_model_layers_43_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/std":0.060546875,"train/train/tensor_act_model_layers_48/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/max_abs":0.00183868408203125,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/max_abs":0.0001926422119140625,"train/train/tensor_param_model_layers_66_mlp_gate_proj_weight/mean":-0.0001468658447265625,"train/train/tensor_act_model_layers_42_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp/std":0.21997335905489115,"train/train/tensor_act_model_layers_61_input_layernorm/norm":5792.604858400335,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/std":6.997882965901281e-05,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/norm":6.46875,"train/train/tensor_act_model_layers_70_self_attn_o_proj/norm":1053.7093850396186,"train/train/layer_model_layers_81/grad/std":9.101319165160758e-05,"train/train/tensor_act_model_layers_59_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_51/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/max_abs":0.000644683837890625,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_73_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/std":5.624620280562229e-05,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/std":9.511640319181486e-05,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/mean":8.96453857421875e-05,"train/train/tensor_act_model_layers_52_input_layernorm/std":1.000000441766945,"train/train/tensor_act_model_layers_46_mlp/norm":369.4766324211434,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/std":8.229631672898992e-05,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn/norm":371.4009388291772,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48/norm":7273.108909965959,"train/train/layer__model_layers_62/param/max_abs":1,"train/train/layer_model_layers_50/act/max_abs":9.8125,"train/train/tensor_act_model_layers_0_self_attn_v_proj/mean":-0.0017910003662109375,"train/train/layer__model_layers_0/param/std":0.051988288889287866,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_mlp_gate_proj_weight/std":0.043212890625,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_69/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/max_abs":0.1708984375,"train/train/tensor_param_model_layers_25_mlp_gate_proj_weight/std":0.02685546875,"train/train/tensor_act_model_layers_24_input_layernorm/mean":-0.0181884765625,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/norm":0.002689586115993126,"train/train/tensor_act_model_layers_42_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_15/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_30/param/mean":0.001572530094807308,"train/train/tensor_act_model_layers_6_self_attn/mean":-0.00030541419982910156,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/norm":0.00869911580566558,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/mean":-2.0908191800117493e-07,"train/train/tensor_grad_model_layers_35_mlp_gate_proj_weight/max_abs":0.00051116943359375,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/std":0.033203125,"train/train/tensor_act_model_layers_13_post_attention_layernorm/mean":-0.030303955078125,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/mean":-2.6106834411621094e-05,"train/train/tensor_act_model_layers_21_self_attn_v_proj/std":0.3500987502678533,"train/train/tensor_act_model_layers_40_self_attn/max_abs":1.2890625,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_7/param/max_abs":1,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/mean":-9.250640869140625e-05,"train/train/tensor_act_model_layers_78/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/std":0.051513671875,"train/train/layer_model_layers_58/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_input_layernorm/mean":0.0026874542236328125,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp/norm":765.7445531230093,"train/train/tensor_act_model_layers_70_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_22/grad/max_abs":0.0010833740234375,"train/train/tensor_act_model_layers_48_post_attention_layernorm/max_abs":6.3125,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_65_self_attn_k_proj/std":0.9531291560007523,"train/train/tensor_act_model_layers_18/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/max_abs":0.0004062652587890625,"train/train/tensor_act_model_layers_12_input_layernorm/norm":5792.615112310835,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/std":4.657616745506599e-05,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/max_abs":0.2373046875,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/max_abs":0.2001953125,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_35/grad/max_abs":0.00119781494140625,"train/train/tensor_param_model_layers_67_mlp_gate_proj_weight/norm":6.84375,"train/train/tensor_grad_model_layers_55_mlp_gate_proj_weight/mean":6.530899554491043e-08,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/max_abs":0.0008087158203125,"train/train/tensor_act_model_layers_86_input_layernorm/max_abs":5.9375,"train/train/tensor_act_model_layers_34_self_attn_o_proj/std":0.07055711968249352,"train/train/layer__model_layers_35/param/max_abs":1,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/std":4.426599722140256e-05,"train/train/tensor_act_model_layers_38_self_attn_v_proj/std":0.39355592595833183,"train/train/layer_model_layers_85/grad/std":7.978473393899303e-05,"train/train/tensor_act_model_layers_63_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12/norm":7446.257792635479,"train/train/tensor_grad_model_layers_46_mlp_gate_proj_weight/std":5.092549834611225e-05,"train/train/layer_model_layers_87/grad/norm":0.07258023464590663,"train/train/tensor_act_model_layers_87_self_attn_q_proj/norm":7088.81547762505,"train/train/layer_model_layers_31/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_70_self_attn_v_proj/max_abs":3.1875,"train/train/layer__model_layers_39/param/mean":0.0014656917911237933,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/mean":0.00012302398681640625,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_mlp_down_proj/std":0.05749526558990268,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/max_abs":0.1611328125,"train/train/tensor_act_model_layers_46_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_69_mlp_gate_proj_weight/std":0.03857421875,"train/train/tensor_act_model_layers_59_self_attn_k_proj/norm":5010.114275213632,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/mean":-9.834766387939453e-06,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/max_abs":0.000339508056640625,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_81/param/norm":24.60867558529918,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/max_abs":0.12353515625,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_64/act/max_abs":9.9375,"train/train/tensor_act_model_layers_68_input_layernorm/max_abs":5.5625,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_14/param/mean":0.001532603723582537,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_input_layernorm/max_abs":6.75,"train/train/tensor_act_model_layers_34_mlp_down_proj/mean":0.0006389617919921875,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp_up_proj/std":0.4511719268637788,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/std":5.8843767114759666e-05,"train/train/tensor_grad_model_layers_23_mlp_gate_proj_weight/mean":1.7462298274040222e-09,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/norm":0.001195031168907436,"train/train/tensor_act_model_layers_28_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/std":9.099742077407067e-05,"train/train/layer__model_layers_7/param/norm":19.60070151805797,"train/train/tensor_act_model_layers_27/std":1.248053323851447,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp_up_proj/std":0.5156251072431944,"train/train/tensor_act_model_layers_74_self_attn/std":0.20703476936681248,"train/train/tensor_act_model_layers_38_self_attn/max_abs":1.4765625,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/mean":-6.435438990592957e-07,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/norm":0.012422603669777097,"train/train/tensor_act_model_layers_19_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp/max_abs":0.64453125,"train/train/tensor_param_model_layers_52_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/mean":-0.00018215179443359375,"train/train/tensor_param_model_layers_24_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_27_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_input_layernorm/mean":0.005573272705078125,"train/train/tensor_act_model_layers_82_mlp_down_proj/mean":0.0010061264038085938,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/max_abs":0.00030517578125,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/std":0.031494140625,"train/train/tensor_act_model_layers_27_self_attn_o_proj/norm":376.8328593210019,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/norm":0.034135408330285215,"train/train/tensor_act_model_layers_51_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/max_abs":0.00022029876708984375,"train/train/tensor_act_model_layers_56_mlp_gate_proj/mean":0.00885009765625,"train/train/layer_model_layers_23/grad/mean":-1.8465903684985063e-08,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/mean":-1.337612047791481e-07,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/norm":0.021402462951473026,"train/train/tensor_act_model_layers_33_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp_down_proj/std":0.17578129506565882,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_mlp_gate_proj_weight/norm":0.01501694328324652,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/max_abs":0.000522613525390625,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_45_post_attention_layernorm/max_abs":6.34375,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/max_abs":0.000431060791015625,"train/train/layer_model_layers_78/grad/max_abs":0.00157928466796875,"train/train/tensor_act_model_layers_7_self_attn_k_proj/norm":4882.123115351203,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/std":0.0005910687004040075,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/std":6.207630242702707e-05,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_49/param/norm":21.77881196003641,"train/train/tensor_act_model_layers_6_mlp_gate_proj/max_abs":2.28125,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_6/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/max_abs":0.130859375,"train/train/tensor_act_model_layers_85_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_o_proj/mean":0.0016803741455078125,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/mean":3.0547380447387695e-06,"train/train/tensor_act_model_layers_58_input_layernorm/norm":5792.6079101586065,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/norm":8.5625,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_68/std":1.441414448093822,"train/train/tensor_act_model_layers_8_self_attn_q_proj/mean":0.0960693359375,"train/train/tensor_param_model_layers_80_mlp_gate_proj_weight/std":0.042724609375,"train/train/tensor_act_model_layers_90_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_gate_proj_weight/std":0.0341796875,"train/train/tensor_param_model_layers_47_mlp_gate_proj_weight/max_abs":0.1728515625,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/norm":3.046875,"train/train/layer__model_layers_21/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/max_abs":0.00049591064453125,"train/train/layer_model_layers_28/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/std":0.0269775390625,"train/train/layer__model_layers_41/param/norm":21.374600233834784,"train/train/tensor_act_model_layers_18_self_attn_q_proj/mean":-0.05078125,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/mean":7.05718994140625e-05,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/mean":-0.0003814697265625,"train/train/tensor_act_model_layers_7_mlp_up_proj/std":0.22070317181337754,"train/train/tensor_act_model_layers_28_mlp_up_proj/max_abs":1.703125,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/max_abs":0.1826171875,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_up_proj/max_abs":1.9609375,"train/train/tensor_act_model_layers_43_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_mlp_gate_proj_weight/std":5.570005308469428e-05,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/norm":0.017192098695920366,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_18_self_attn_v_proj/norm":1687.129866910355,"train/train/tensor_act_model_layers_83_self_attn_k_proj/std":0.9091815968953374,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/norm":6.59375,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/std":0.044921875,"train/train/tensor_act_model_layers_7_self_attn/mean":-0.0006208419799804688,"train/train/tensor_act_model_layers_80_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0/norm":7449.449360608512,"train/train/layer_model_layers_24/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/norm":0.02373718355097468,"train/train/tensor_act_model_layers_40_mlp_up_proj/norm":2562.152576532961,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_42_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn/mean":0.001911163330078125,"train/train/tensor_act_model_layers_27_mlp_down_proj/std":0.03985612114906112,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/max_abs":0.0002841949462890625,"train/train/tensor_act_model_layers_67/frac_near_dtype_limit":0,"train/train/layer__model_layers_19/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_post_attention_layernorm/mean":0.00669097900390625,"train/train/layer_model_layers_34/grad/std":4.7150373926070575e-05,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/norm":5.6875,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_77/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/std":0.03369140625,"train/train/tensor_param_model_layers_58_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/max_abs":0.0006561279296875,"train/train/tensor_act_model_layers_53/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_post_attention_layernorm/norm":5792.61242675935,"train/train/tensor_act_model_layers_80_self_attn_o_proj/mean":0.000904083251953125,"train/train/tensor_param_model_layers_41_mlp_gate_proj_weight/std":0.03076171875,"train/train/tensor_act_model_layers_9/mean":-0.03265380859375,"train/train/tensor_act_model_layers_81/norm":10421.819471605215,"train/train/tensor_act_model_layers_75_self_attn/max_abs":1.671875,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/mean":2.2258609533309937e-07,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/std":4.288140490387654e-05,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/norm":7.84375,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/mean":1.5929341316223145e-05,"train/train/tensor_act_model_layers_15_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn/max_abs":0.8671875,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_mlp_gate_proj_weight/mean":0.0005340576171875,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/std":0.0260009765625,"train/train/tensor_act_model_layers_32_self_attn_o_proj/mean":0.0005407333374023438,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_input_layernorm/mean":0.0008220672607421875,"train/train/tensor_act_model_layers_19_self_attn/std":0.08364283474223369,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_mlp_gate_proj/std":0.43505942299632155,"train/train/layer_model_layers_72/grad/mean":-1.1114484969055782e-08,"train/train/tensor_act_model_layers_74_post_attention_layernorm/mean":0.007213592529296875,"train/train/tensor_act_model_layers_21_self_attn_q_proj/std":1.0957083931454576,"train/train/tensor_act_model_layers_66_mlp_gate_proj/std":0.43164080273508065,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_57_mlp_up_proj/norm":3178.882516118502,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/norm":0.0029781036404707326,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/std":0.0001260452546389458,"train/train/tensor_grad_model_layers_75_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/max_abs":0.00118255615234375,"train/train/tensor_param_model_layers_15_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/std":0.026611328125,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_67_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_input_layernorm/mean":0.0020270347595214844,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/std":0.0001058384809804918,"train/train/tensor_act_model_layers_23_mlp/norm":221.10912251431404,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/mean":4.212051862850785e-08,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/norm":7.8125,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/max_abs":0.00011348724365234375,"train/train/tensor_act_model_layers_29_input_layernorm/std":1.0000011104851383,"train/train/tensor_param_model_layers_28_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/norm":0.016075002111209356,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/std":0.0252685546875,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_34_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_input_layernorm/norm":5792.606079102483,"train/train/tensor_act_model_layers_14_self_attn_k_proj/std":0.9150406807933,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/std":0.05615234375,"train/train/tensor_param_model_layers_45_mlp_gate_proj_weight/std":0.03125,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/norm":0.01629764394875328,"train/train/tensor_grad_model_embed_tokens_weight/max_abs":0.0128173828125,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/max_abs":0.000667572021484375,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/std":0.0242919921875,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_64_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/norm":0.03381413758661188,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/mean":-0.0002593994140625,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/max_abs":0.1982421875,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/max_abs":0.00131988525390625,"train/train/tensor_act_model_layers_68_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/max_abs":0.00107574462890625,"train/train/tensor_act_model_layers_27_input_layernorm/norm":5792.616455080127,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/max_abs":0.0002384185791015625,"train/train/tensor_param_model_layers_75_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/mean":2.4121254682540894e-07,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/mean":6.429851055145264e-06,"train/train/tensor_act_model_layers_8_mlp_gate_proj/max_abs":1.9609375,"train/train/layer__model_layers_83/param/max_abs":1,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_gate_proj_weight/max_abs":0.16015625,"train/train/tensor_act_model_layers_38_self_attn_q_proj/std":0.9628930909123289,"train/train/layer_model_layers_91/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_29/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/max_abs":0.00032806396484375,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/mean":-1.3760291039943695e-07,"train/train/tensor_act_model_layers_90_mlp/norm":2434.5026698431375,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn/mean":6.048381328582764e-05,"train/train/tensor_param_model_layers_18_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_33/param/norm":20.57575797889959,"train/train/tensor_act_model_layers_74_mlp_gate_proj/norm":3794.8041436185727,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/norm":0.032844854479656245,"train/train/tensor_act_model_layers_21_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_65/act/mean":0.002581732613699777,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/max_abs":0.1806640625,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/norm":0.00467828102342168,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/norm":3.765625,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/norm":4.875,"train/train/layer__model_layers_84/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_post_attention_layernorm/max_abs":6.125,"train/train/tensor_param_model_layers_75_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_4_input_layernorm/mean":-0.031982421875,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp_gate_proj/max_abs":1.796875,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/norm":0.03390193894222959,"train/train/tensor_act_model_layers_86_mlp_down_proj/mean":-0.002292633056640625,"train/train/tensor_act_model_layers_5_mlp/norm":273.00844885139077,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_k_proj/std":0.9072324318382844,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/max_abs":0.1376953125,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/norm":0.0017626749774573563,"train/train/tensor_act_model_layers_62_self_attn_k_proj/max_abs":4.9375,"train/train/layer_model_layers_84/act/norm":18317.73615646196,"train/train/tensor_act_model_layers_28_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27/max_abs":8.875,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/std":7.570687536426084e-05,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/mean":0.000247955322265625,"train/train/tensor_grad_model_layers_43_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85/mean":0.0099945068359375,"train/train/tensor_act_model_layers_56_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/std":0.0281982421875,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/max_abs":0.0004863739013671875,"train/train/layer__model_layers_44/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_mlp_gate_proj/max_abs":2.1875,"train/train/tensor_act_model_layers_50_self_attn_q_proj/norm":5093.698167611493,"train/train/layer__model_layers_63/param/std":0.05543814136431932,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_mlp_gate_proj_weight/max_abs":0.1259765625,"train/train/tensor_act_model_layers_5/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/mean":-7.057678885757923e-08,"train/train/tensor_act_model_layers_23_mlp_gate_proj/max_abs":2.21875,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_45_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_k_proj/max_abs":4.65625,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp_up_proj/norm":2281.936837564336,"train/train/layer__model_layers_65/param/mean":0.0016310955172581904,"train/train/tensor_act_model_layers_14_mlp_up_proj/mean":6.455183029174805e-05,"train/train/tensor_act_model_layers_18_mlp_down_proj/std":0.03790298349009305,"train/train/layer__model_layers_31/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/max_abs":0.001739501953125,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/norm":0.0012577803719457156,"train/train/tensor_param_model_layers_27_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_29_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_0/act/norm":17972.328197801486,"train/train/tensor_param_model_layers_45_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/std":6.862143048196147e-05,"train/train/layer_model_layers_47/act/mean":-0.0029757129294531687,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_41/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/std":1.9488268583102038e-05,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_post_attention_layernorm/mean":-0.030303955078125,"train/train/tensor_param_model_layers_63_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/norm":0.0021622570938469666,"train/train/tensor_act_model_layers_59_self_attn/std":0.1399052024705307,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/norm":0.013223164828956706,"train/train/layer_model_layers_85/grad/norm":0.06459562306636613,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/max_abs":0.13671875,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/std":1.782415895013673e-05,"train/train/tensor_act_model_layers_8_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/std":3.5726865563621316e-05,"train/train/tensor_grad_model_layers_3_mlp_gate_proj_weight/max_abs":0.000751495361328125,"train/train/layer__model_layers_87/param/std":0.061997244120444536,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/max_abs":0.123046875,"train/train/layer_model_layers_53/act/max_abs":10,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/mean":-0.000247955322265625,"train/train/tensor_grad_model_layers_78_mlp_gate_proj_weight/norm":0.029058247531982627,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/max_abs":0.00048828125,"train/train/layer__model_layers_70/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/norm":3.4375,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/norm":0.005342189000328368,"train/train/tensor_act_model_layers_3_self_attn_o_proj/mean":-0.0005626678466796875,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/max_abs":0.00021648406982421875,"train/train/tensor_act_model_layers_11_post_attention_layernorm/max_abs":5.65625,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_6_mlp/std":0.07031589295263843,"train/train/tensor_act_model_layers_60_mlp_up_proj/mean":0.002288818359375,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_40/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/norm":4.9375,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/norm":0.0014094880527365599,"train/train/tensor_act_model_layers_35_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/std":7.953227330104321e-05,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/norm":0.0047122243670340595,"train/train/tensor_param_model_layers_18_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/std":0.00011892015149833045,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/norm":0.003938689243392232,"train/train/tensor_act_model_layers_65_post_attention_layernorm/mean":0.00200653076171875,"train/train/layer_model_layers_87/grad/max_abs":0.0021820068359375,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/max_abs":0.000396728515625,"train/train/layer__model_layers_20/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_mlp/max_abs":0.65625,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/std":0.026123046875,"train/train/layer__model_layers_28/param/std":0.05169347928780027,"train/train/tensor_param_model_layers_76_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/norm":0.011694326380843237,"train/train/tensor_act_model_layers_23_self_attn_o_proj/std":0.04907457551704344,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/std":0.0576171875,"train/train/tensor_act_model_layers_35_mlp_up_proj/std":0.28906251014386464,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/mean":7.655471563339233e-06,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/std":0.0303955078125,"train/train/tensor_act_model_layers_61_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/std":0.0262451171875,"train/train/tensor_act_model_layers_33_mlp_up_proj/max_abs":1.9375,"train/train/tensor_act_model_layers_43_mlp_down_proj/std":0.05767833230516724,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/norm":5.78125,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/mean":0.000274658203125,"train/train/tensor_act_model_layers_25_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/max_abs":0.0011138916015625,"train/train/tensor_act_model_layers_24_post_attention_layernorm/mean":-0.01837158203125,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_mlp_down_proj/mean":0.0007076263427734375,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/std":0.0267333984375,"train/train/tensor_act_model_layers_70_self_attn/mean":0.0009250640869140625,"train/train/layer_model_layers_0/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_28_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp_down_proj/mean":7.265806198120117e-05,"train/train/tensor_act_model_layers_38_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/norm":0.021734858240806527,"train/train/tensor_act_model_layers_93_mlp_gate_proj/norm":7078.983735947896,"train/train/tensor_param_model_layers_83_mlp_gate_proj_weight/norm":7.875,"train/train/tensor_act_model_layers_16_mlp_down_proj/std":0.03564495862243693,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/std":6.212669925583174e-05,"train/train/tensor_act_model_layers_8_mlp/norm":239.61430995911246,"train/train/tensor_act_model_layers_80_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_gate_proj/mean":-0.00144195556640625,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/norm":3.96875,"train/train/layer_model_layers_90/grad/mean":-1.0826443314738281e-07,"train/train/tensor_act_model_layers_58_self_attn_q_proj/mean":0.051025390625,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/std":0.0400390625,"train/train/tensor_act_model_layers_51_mlp_up_proj/max_abs":2.109375,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/max_abs":0.11669921875,"train/train/tensor_act_model_layers_57_self_attn_o_proj/std":0.13110669383489126,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/std":3.8944526252385465e-05,"train/train/tensor_act_model_layers_83_mlp_up_proj/max_abs":3.28125,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/std":0.04736328125,"train/train/tensor_act_model_layers_73_mlp_down_proj/max_abs":1,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/norm":0.02253230821767225,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/norm":10.125,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/max_abs":0.1533203125,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_up_proj/mean":0.0018444061279296875,"train/train/tensor_act_model_layers_33_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_79/grad/norm":0.062494550943324144,"train/train/tensor_act_model_layers_20_self_attn_o_proj/max_abs":0.66015625,"train/train/tensor_act_model_layers_84_mlp_up_proj/norm":4468.1338956292775,"train/train/tensor_act_model_layers_41_input_layernorm/max_abs":6.71875,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/mean":-0.000324249267578125,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/norm":0.0007208188199593886,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/std":2.6819198469072023e-05,"train/train/tensor_act_model_layers_36_mlp_gate_proj/mean":-0.00429534912109375,"train/train/tensor_act_model_layers_58_mlp_down_proj/norm":532.4672282305404,"train/train/tensor_act_model_layers_52_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/max_abs":0.0001678466796875,"train/train/tensor_act_model_layers_65_mlp_gate_proj/max_abs":2.484375,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/mean":1.4487653970718384e-05,"train/train/layer__model_layers_85/param/norm":24.571980517705526,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/mean":-4.9174559535458684e-08,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/mean":-0.00018215179443359375,"train/train/tensor_act_model_layers_89_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_48/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_o_proj/norm":371.4009388291772,"train/train/tensor_param_model_layers_1_mlp_gate_proj_weight/max_abs":0.1572265625,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/norm":0.029952159525753648,"train/train/layer_model_layers_80/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/mean":0.0003147125244140625,"train/train/tensor_act_model_layers_21_post_attention_layernorm/std":1.00000161444638,"train/train/tensor_act_model_layers_65_self_attn_k_proj/norm":5524.479667337834,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/std":4.960780308797584e-05,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/max_abs":0.1171875,"train/train/tensor_param_model_layers_55_mlp_gate_proj_weight/std":0.03369140625,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/max_abs":0.000644683837890625,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_63/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/max_abs":0.1435546875,"train/train/layer_model_layers_13/act/norm":14126.006518373684,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/mean":-0.00010585784912109375,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/std":0.0213623046875,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/mean":-2.014636993408203e-05,"train/train/tensor_act_model_layers_13_mlp_up_proj/norm":1945.065926418733,"train/train/layer_model_layers_6/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/norm":0.01897525976679211,"train/train/tensor_act_model_layers_18_mlp_up_proj/norm":2003.1670552942583,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_self_attn_v_proj/norm":2255.9870748397234,"train/train/tensor_act_model_layers_46_self_attn_q_proj/std":0.9697320190567186,"train/train/tensor_act_model_layers_82_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp_gate_proj/mean":0.00463104248046875,"train/train/layer_model_layers_3/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_input_layernorm/std":1.0000008431667167,"train/train/tensor_act_model_layers_85_mlp_down_proj/norm":1446.4339918460219,"train/train/tensor_act_model_layers_90_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_10_mlp_down_proj/mean":-0.0002684593200683594,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/std":0.00012774031356161834,"train/train/layer_model_layers_92/grad/max_abs":0.00080108642578125,"train/train/tensor_act_model_layers_63_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_gate_proj/max_abs":4.125,"train/train/tensor_act_model_layers_23_mlp_down_proj/std":0.03814712066621676,"train/train/tensor_act_model_layers_71/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/max_abs":0.000202178955078125,"train/train/tensor_act_model_layers_49_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_mlp_gate_proj_weight/mean":1.986045390367508e-07,"train/train/tensor_act_model_layers_56_mlp_down_proj/std":0.08483915448218864,"train/train/tensor_act_model_layers_46_input_layernorm/mean":-0.010986328125,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/std":0.03076171875,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/max_abs":0.23046875,"train/train/tensor_act_model_layers_14_mlp_down_proj/norm":206.20778720477077,"train/train/tensor_act_model_layers_68_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn/max_abs":1.4140625,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/mean":-7.343292236328125e-05,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_gate_proj/std":0.46338002473075146,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/std":0.033203125,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/std":0.039306640625,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/std":0.049072265625,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/std":8.722171972028536e-05,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_mlp/mean":0.0008287429809570312,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/mean":0.00034332275390625,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/max_abs":0.0002498626708984375,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/std":2.2992505773221084e-05,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/std":1.8878469228816417e-05,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_90_post_attention_layernorm/mean":0.00897979736328125,"train/train/tensor_param_model_embed_tokens_weight/max_abs":0.5,"train/train/tensor_act_model_layers_2_mlp_down_proj/std":0.18408398475419915,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn/frac_near_dtype_limit":0,"train/train/layer__model_layers_36/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/mean":-1.8463470041751862e-07,"train/train/tensor_param_model_layers_15_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/mean":-0.00013256072998046875,"train/train/tensor_act_model_layers_61_self_attn/norm":714.5199832580456,"train/train/tensor_act_model_layers_16_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_mlp_gate_proj_weight/mean":-6.416812539100647e-07,"train/train/tensor_act_model_layers_35_mlp_up_proj/norm":2361.9533678745847,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/max_abs":0.19921875,"train/train/tensor_act_model_layers_55_self_attn_q_proj/max_abs":5.125,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/max_abs":0.134765625,"train/train/tensor_act_model_layers_37_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn_o_proj/mean":0.0002925395965576172,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp/max_abs":1.046875,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_mlp_gate_proj_weight/mean":-0.0002307891845703125,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/norm":0.0027655266041143297,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/max_abs":0.19921875,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/std":0.031494140625,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/std":0.048095703125,"train/train/tensor_param_model_layers_0_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/mean":0.0001373291015625,"train/train/tensor_act_model_layers_10_input_layernorm/norm":5792.60546875283,"train/train/tensor_act_model_layers_62_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_mlp_gate_proj/mean":0.004241943359375,"train/train/tensor_act_model_layers_27_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn_o_proj/max_abs":1.5,"train/train/tensor_act_model_layers_38_mlp_down_proj/mean":0.0003247261047363281,"train/train/tensor_act_model_layers_43_self_attn_o_proj/max_abs":1.5546875,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/std":0.043212890625,"train/train/tensor_act_model_layers_57_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_q_proj/std":0.9902368488835764,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/norm":5.8125,"train/train/tensor_act_model_layers_62_post_attention_layernorm/norm":5792.60266113721,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/std":1.4928334126483654e-05,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/max_abs":0.00067901611328125,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/max_abs":0.00013446807861328125,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/mean":0.000782012939453125,"train/train/tensor_act_model_layers_56_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_gate_proj/std":0.2915052560874063,"train/train/tensor_act_model_layers_90_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_input_layernorm/std":1.0000012312076856,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/mean":9.388895705342293e-08,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/mean":5.46453520655632e-07,"train/train/layer_model_layers_79/grad/max_abs":0.001190185546875,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/max_abs":0.21484375,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/std":0.031494140625,"train/train/tensor_act_model_layers_60_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_input_layernorm/std":1.0000003123422596,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/mean":8.869171142578125e-05,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/mean":-1.1816155165433884e-07,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/mean":2.9546208679676056e-07,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/max_abs":0.177734375,"train/train/tensor_param_model_layers_18_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_37/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_input_layernorm/std":1.0000005911186307,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/mean":6.686896085739136e-06,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/mean":3.266334533691406e-05,"train/train/layer_model_layers_78/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/max_abs":0.0003662109375,"train/train/tensor_act_model_layers_92_mlp/max_abs":5.84375,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/std":3.1257301710477825e-05,"train/train/tensor_act_model_layers_8_self_attn_k_proj/norm":5540.30003848565,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/norm":0.01826338734882947,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/mean":-1.5330442693084478e-07,"train/train/tensor_act_model_layers_35_self_attn_k_proj/mean":0.025115966796875,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/std":0.052734375,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/mean":2.0625884644687176e-07,"train/train/tensor_act_model_layers_67_self_attn_k_proj/std":0.8115253064011263,"train/train/layer_model_layers_12/act/std":0.6368311060675862,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/std":0.0537109375,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/norm":5.5,"train/train/tensor_act_model_layers_58_mlp/max_abs":0.6640625,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/mean":-0.022613525390625,"train/train/tensor_param_model_layers_9_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_mlp_gate_proj_weight/std":0.0257568359375,"train/train/tensor_act_model_layers_34_input_layernorm/mean":-0.0119476318359375,"train/train/tensor_act_model_layers_6_self_attn/max_abs":0.65234375,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/std":0.0296630859375,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/mean":4.1961669921875e-05,"train/train/tensor_act_model_layers_36_mlp_down_proj/norm":292.7850369317611,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_q_proj/std":0.8349627221517842,"train/train/tensor_act_model_layers_91_mlp_up_proj/mean":0.026519775390625,"train/train/layer__model_layers_11/param/max_abs":1,"train/train/tensor_act_model_layers_58_self_attn_k_proj/max_abs":5.875,"train/train/tensor_act_model_layers_14/mean":-0.031951904296875,"train/train/tensor_act_model_layers_58_mlp_gate_proj/mean":-0.006591796875,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/std":3.43665787331008e-05,"train/train/layer_model_layers_74/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_41/act/max_abs":9.375,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_gate_proj_weight/std":3.8939648100898856e-05,"train/train/tensor_act_model_embed_tokens/max_abs":0.5,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/max_abs":0.0014495849609375,"train/train/tensor_act_model_layers_66_self_attn_v_proj/max_abs":3.84375,"train/train/tensor_act_model_layers_85_post_attention_layernorm/mean":0.00675201416015625,"train/train/tensor_param_model_layers_50_mlp_gate_proj_weight/norm":5.90625,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_80/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_86/act/mean":-0.004500389099121094,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/norm":5.6875,"train/train/tensor_act_model_layers_63_mlp_down_proj/norm":617.7870798821212,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/max_abs":0.1484375,"train/train/tensor_act_model_layers_0_mlp_up_proj/max_abs":5.84375,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp_down_proj/norm":580.2673105401778,"train/train/tensor_act_model_layers_28_self_attn_o_proj/max_abs":1.125,"train/train/tensor_act_model_layers_79_self_attn_k_proj/max_abs":5.6875,"train/train/tensor_param_model_layers_85_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/std":0.044921875,"train/train/tensor_act_model_layers_85_mlp_gate_proj/mean":-0.0131378173828125,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/max_abs":0.00022220611572265625,"train/train/tensor_act_model_layers_1_input_layernorm/max_abs":4.9375,"train/train/global/param/norm":231.52526691216664,"train/train/tensor_act_model_layers_85_self_attn/mean":0.0012416839599609375,"train/train/tensor_act_model_layers_8_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_o_proj/mean":-6.103515625e-05,"train/train/tensor_act_model_layers_52_mlp_down_proj/max_abs":0.51953125,"train/train/tensor_act_model_layers_21_self_attn/norm":485.13577003861036,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn/std":0.2448744342602649,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/norm":9.0625,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn/mean":0.00015103816986083984,"train/train/tensor_act_model_layers_21_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp_up_proj/max_abs":3.03125,"train/train/layer__model_layers_93/param/mean":0.0017411682795436818,"train/train/tensor_act_model_layers_32_post_attention_layernorm/std":1.000000767875168,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/max_abs":0.0001163482666015625,"train/train/tensor_act_model_layers_52_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/mean":6.389617919921875e-05,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_50_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_mlp_gate_proj_weight/std":0.0311279296875,"train/train/tensor_param_model_layers_5_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_mlp/max_abs":1.078125,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn_v_proj/mean":-0.0004476308822631836,"train/train/layer_model_layers_32/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/mean":-1.9080471247434616e-07,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/max_abs":0.00022125244140625,"train/train/layer_model_layers_30/grad/norm":0.03635399979528492,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/mean":0.00019073486328125,"train/train/tensor_param_model_layers_39_mlp_gate_proj_weight/norm":5.53125,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/max_abs":0.001007080078125,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_o_proj/norm":1389.7820880256709,"train/train/tensor_grad_model_layers_37_mlp_gate_proj_weight/norm":0.0179507704635862,"train/train/tensor_act_model_layers_29_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_k_proj/std":0.9492230379680828,"train/train/tensor_act_model_layers_42_self_attn_o_proj/mean":-0.0007524490356445312,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/std":4.8833315299138745e-05,"train/train/tensor_act_model_layers_19_input_layernorm/mean":-0.02618408203125,"train/train/tensor_act_model_layers_81_self_attn_o_proj/norm":1419.2259170283617,"train/train/tensor_grad_model_embed_tokens_weight/norm":0.1917242405982702,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/norm":0.001067317937552931,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/mean":-4.526373231783509e-08,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/std":4.090914931349013e-05,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/mean":-1.0803341865539551e-07,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp_gate_proj/mean":0.0017147064208984375,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/norm":0.0015826856693086369,"train/train/tensor_act_model_layers_14_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/std":4.352673762563217e-05,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_33/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_down_proj/mean":-0.0005235671997070312,"train/train/tensor_act_model_layers_67_mlp_up_proj/std":0.4296877234623271,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/norm":0.0014395618248869075,"train/train/layer__model_layers_33/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_25/grad/mean":-1.240358942198493e-07,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_k_proj/std":0.8828126197367562,"train/train/tensor_act_model_layers_14_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_36/grad/max_abs":0.0015106201171875,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/max_abs":0.001373291015625,"train/train/tensor_param_model_layers_47_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/max_abs":0.12451171875,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn/max_abs":0.734375,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_post_attention_layernorm/norm":5792.609497070748,"train/train/tensor_param_model_layers_20_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp_up_proj/std":0.3081066489563096,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/norm":0.005582423338523856,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/std":4.82339674632091e-05,"train/train/tensor_act_model_layers_60_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/mean":0.00022125244140625,"train/train/layer_model_layers_36/act/norm":14021.348750063646,"train/train/tensor_act_model_layers_52_post_attention_layernorm/std":1.0000004173197379,"train/train/tensor_act_model_layers_80_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_gate_proj/norm":2717.245177691576,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/mean":4.2689498513936996e-07,"train/train/tensor_act_model_layers_91_post_attention_layernorm/mean":0.01275634765625,"train/train/tensor_param_model_layers_66_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_62_mlp_gate_proj/std":0.3984375107376013,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/mean":-1.1222437024116516e-07,"train/train/tensor_act_model_layers_60_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_mlp/norm":2141.399245173063,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/mean":-0.00012683868408203125,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_self_attn_o_proj/max_abs":2.46875,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_gate_proj/norm":2061.0471050009464,"train/train/layer_model_layers_30/act/norm":13395.78485854265,"train/train/tensor_grad_model_layers_59_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_gate_proj/mean":-0.004093170166015625,"train/train/tensor_act_model_layers_33_self_attn_q_proj/norm":5444.505506477494,"train/train/tensor_act_model_layers_42_mlp/norm":345.73317908750266,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/norm":6.875,"train/train/tensor_act_model_layers_2_self_attn_o_proj/norm":224.34503850420992,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/max_abs":0.0004425048828125,"train/train/tensor_act_model_layers_49_self_attn_k_proj/mean":-0.0089874267578125,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_80_input_layernorm/norm":5792.609741216415,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36/mean":-0.0163116455078125,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/std":8.8976658993002e-05,"train/train/tensor_act_model_layers_78_self_attn_q_proj/norm":6375.580853704999,"train/train/tensor_act_model_layers_33_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_v_proj/max_abs":2.328125,"train/train/layer_model_layers_71/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45/std":1.2421880355982255,"train/train/layer__model_layers_5/param/frac_near_user_limit":0,"train/train/layer_model_layers_54/grad/max_abs":0.0015411376953125,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/mean":5.532056093215942e-06,"train/train/tensor_act_model_layers_74_post_attention_layernorm/norm":5792.606445314712,"train/train/tensor_grad_model_layers_93_mlp_gate_proj_weight/norm":0.03976027027343365,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/max_abs":8.869171142578125e-05,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/max_abs":0.0004329681396484375,"train/train/tensor_act_model_layers_47_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_k_proj/max_abs":5.21875,"train/train/tensor_act_model_layers_8_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/mean":0.00048828125,"train/train/tensor_act_model_layers_34_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/max_abs":0.000331878662109375,"train/train/tensor_act_model_layers_75_mlp_up_proj/std":0.46484398741675975,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/std":7.79415593543823e-05,"train/train/layer_model_layers_10/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/mean":-0.000301361083984375,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp_up_proj/norm":2520.821039524641,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/max_abs":0.000637054443359375,"train/train/layer__model_layers_77/param/std":0.06019171536182161,"train/train/tensor_act_model_layers_59_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/std":3.782898059268003e-05,"train/train/tensor_act_model_layers_88_self_attn/mean":0.003040313720703125,"train/train/layer__model_layers_89/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/max_abs":0.1806640625,"train/train/tensor_act_model_layers_17_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_63/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp_gate_proj/max_abs":2.515625,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/max_abs":0.000324249267578125,"train/train/tensor_param_model_layers_4_mlp_gate_proj_weight/norm":4.3125,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/norm":0.015128379493484449,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/mean":-2.547167241573334e-07,"train/train/tensor_act_model_layers_38_self_attn/mean":-2.3156404495239258e-05,"train/train/tensor_act_model_layers_24_self_attn_o_proj/mean":-0.00015419721603393555,"train/train/tensor_act_model_layers_88_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_30/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_mlp_gate_proj_weight/std":4.149074351454606e-05,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/mean":0.0002880096435546875,"train/train/layer__model_layers_59/param/norm":22.731126271315727,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/norm":4.96875,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/mean":-3.6139972507953644e-06,"train/train/tensor_act_model_layers_81_mlp_down_proj/std":0.20678761290597703,"train/train/layer_model_layers_70/grad/mean":-1.3366422228015343e-07,"train/train/tensor_param_model_layers_45_input_layernorm_weight/std":0.0003452301025390625,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/max_abs":0.1796875,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/std":5.865552554685485e-05,"train/train/tensor_act_model_layers_76_mlp_down_proj/max_abs":1.1171875,"train/train/layer_model_layers_60/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/max_abs":0.1865234375,"train/train/tensor_param_model_layers_62_mlp_gate_proj_weight/std":0.0361328125,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/mean":1.8579885363578796e-07,"train/train/tensor_act_model_layers_54/std":1.2890629576010324,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_27/param/std":0.05068962096738881,"train/train/tensor_act_model_layers_75_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_k_proj/max_abs":4.71875,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/max_abs":0.1484375,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/std":0.034912109375,"train/train/layer__model_layers_78/param/std":0.059264067276921,"train/train/layer_model_layers_87/act/mean":0.0011681147984095982,"train/train/tensor_grad_model_layers_50_mlp_gate_proj_weight/mean":1.885928213596344e-08,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/global/act/max_abs":26.125,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/norm":0.0005820695545245871,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/norm":0.005067325304638328,"train/train/tensor_act_model_layers_76_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_32/grad/max_abs":0.00225830078125,"train/train/tensor_act_model_layers_36_mlp_up_proj/std":0.2949218926704576,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/norm":3.28125,"train/train/tensor_grad_model_layers_14_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_up_proj/norm":3139.5696902890118,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/mean":1.2794043868780136e-07,"train/train/tensor_act_model_layers_3_post_attention_layernorm/mean":-0.02642822265625,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_mlp_gate_proj_weight/max_abs":0.16796875,"train/train/tensor_act_model_layers_50_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/mean":-2.889428287744522e-07,"train/train/layer__model_layers_49/param/std":0.05377292628271868,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/std":0.037109375,"train/train/layer_model_layers_83/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/mean":-0.00020599365234375,"train/train/tensor_act_model_layers_62_self_attn/max_abs":1.6484375,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/max_abs":0.0002307891845703125,"train/train/tensor_act_model_layers_86_self_attn_q_proj/norm":6961.847050303357,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/norm":0.01732527823149079,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/max_abs":0.21484375,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/mean":0.00074005126953125,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/mean":-1.2263655662536621e-05,"train/train/layer__model_layers_83/param/mean":0.001666969144586096,"train/train/tensor_param_model_layers_51_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_q_proj/norm":5243.794187481513,"train/train/tensor_act_model_layers_12_post_attention_layernorm/norm":5792.607788093113,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/mean":-8.479692041873932e-07,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/norm":0.014656509718393345,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/mean":0.000141143798828125,"train/train/tensor_act_model_layers_54_self_attn_v_proj/norm":2427.806444828038,"train/train/tensor_act_model_layers_92_self_attn_k_proj/max_abs":5.84375,"train/train/tensor_act_model_layers_89_mlp_up_proj/norm":5185.185717057704,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/mean":-5.6743621826171875e-05,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp_down_proj/norm":237.52016336661853,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/std":0.050048828125,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/max_abs":0.00057220458984375,"train/train/tensor_act_model_layers_54_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/norm":3.546875,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_68/grad/std":9.137771638995356e-05,"train/train/layer_model_layers_12/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_75/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn/norm":193.78448408511994,"train/train/tensor_act_model_layers_69_mlp/max_abs":0.94140625,"train/train/tensor_act_model_layers_41_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/mean":-5.888938903808594e-05,"train/train/tensor_act_model_layers_28_self_attn_v_proj/mean":0.003658294677734375,"train/train/tensor_act_model_layers_56/max_abs":9.875,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_mlp/mean":0.0002837181091308594,"train/train/tensor_act_model_layers_62_self_attn_q_proj/norm":5101.454986254306,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/norm":0.0029696775072957896,"train/train/tensor_act_model_layers_65_post_attention_layernorm/norm":5792.608276369269,"train/train/tensor_act_model_layers_79_input_layernorm/max_abs":5.84375,"train/train/tensor_param_model_layers_6_mlp_gate_proj_weight/std":0.0244140625,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_87_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/mean":-1.0952353477478027e-06,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_v_proj/max_abs":1.3359375,"train/train/layer__model_layers_70/param/max_abs":1,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/max_abs":0.0022735595703125,"train/train/tensor_act_model_layers_45_mlp_gate_proj/norm":2647.5564647660317,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/std":0.031494140625,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_q_proj/mean":0.03125,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/mean":-6.915070116519928e-08,"train/train/tensor_act_model_layers_66/std":1.3945382157745816,"train/train/layer_model_layers_71/act/mean":0.006601401737758091,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_32/act/norm":13639.438315806256,"train/train/tensor_param_model_layers_4_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_73_self_attn_o_proj/norm":711.6674224229876,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6/std":1.291032249990003,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_post_attention_layernorm/norm":5792.607910159388,"train/train/layer__model_layers_57/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/norm":0.015069720984874342,"train/train/tensor_act_model_layers_44_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_68/grad/mean":-9.730795337331834e-08,"train/train/tensor_act_model_layers_8_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/max_abs":0.00023555755615234375,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/max_abs":0.15234375,"train/train/tensor_param_model_layers_51_mlp_gate_proj_weight/std":0.03271484375,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_79_mlp_up_proj/norm":4051.692234586883,"train/train/tensor_act_model_layers_75/norm":9155.662422820269,"train/train/tensor_act_model_layers_70_mlp_gate_proj/std":0.44531259112255206,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_gate_proj/mean":0.0080108642578125,"train/train/tensor_act_model_embed_tokens/norm":669.224928684004,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/std":2.6352475432107014e-05,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/mean":0.00011348724365234375,"train/train/tensor_act_model_layers_93/max_abs":26.125,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/norm":7.0625,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/mean":-5.037873052060604e-08,"train/train/tensor_param_model_layers_76_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/mean":2.601009327918291e-07,"train/train/layer_model_layers_43/grad/std":4.907727560040858e-05,"train/train/tensor_act_model_layers_93_self_attn_q_proj/std":1.1289130675958072,"train/train/tensor_act_model_layers_29_self_attn_o_proj/mean":-0.0005130767822265625,"train/train/tensor_param_model_layers_83_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp_up_proj/norm":4027.1812360348167,"train/train/tensor_act_model_layers_48_self_attn_k_proj/mean":-0.019439697265625,"train/train/tensor_act_model_layers_19_mlp/mean":0.0011043548583984375,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_v_proj/mean":-0.00667572021484375,"train/train/tensor_grad_model_layers_6_mlp_gate_proj_weight/max_abs":0.00102996826171875,"train/train/tensor_act_model_layers_83_post_attention_layernorm/max_abs":5.875,"train/train/layer_model_layers_74/grad/mean":-2.0347291519414987e-07,"train/train/tensor_param_model_layers_58_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_43/max_abs":9.3125,"train/train/tensor_act_model_layers_23_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/mean":-0.00032806396484375,"train/train/tensor_act_model_layers_26_mlp_up_proj/std":0.2578125071330842,"train/train/tensor_act_model_layers_33_self_attn_v_proj/mean":-0.0009613037109375,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/std":2.2692940678438622e-05,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/max_abs":0.00014019012451171875,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/norm":5.71875,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/max_abs":0.1142578125,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_o_proj/mean":0.0005002021789550781,"train/train/layer_model_layers_27/act/norm":13721.48003729493,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_22_mlp/std":0.03631607668455286,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/mean":-2.773595042526722e-08,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_2/param/mean":0.001449632570263748,"train/train/tensor_act_/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn/std":0.07190223005910458,"train/train/tensor_param_model_layers_88_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/mean":-0.000308990478515625,"train/train/tensor_param_model_layers_44_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/norm":0.003607711494315585,"train/train/tensor_act_model_layers_69_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_54/grad/std":6.359924132013044e-05,"train/train/tensor_act_model_layers_43_mlp_up_proj/mean":-0.0146026611328125,"train/train/tensor_act_model_layers_62_mlp_gate_proj/max_abs":2.203125,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/std":4.1567569390201536e-05,"train/train/tensor_act_model_layers_6_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83/norm":10898.40812940153,"train/train/tensor_act_model_layers_44_self_attn_o_proj/std":0.07934697269377686,"train/train/tensor_act_model_layers_57_post_attention_layernorm/std":1.0000007509661963,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_88_self_attn_v_proj/std":0.4604523356736461,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/std":0.031005859375,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/std":1.902578862185198e-05,"train/train/layer_model_layers_73/grad/std":7.804500646201765e-05,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/norm":3.875,"train/train/tensor_act_model_layers_63_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp_gate_proj/mean":0.003025054931640625,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_14_mlp_gate_proj_weight/norm":0.014226364046374888,"train/train/tensor_act_model_layers_8_self_attn_k_proj/max_abs":4.625,"train/train/layer_model_layers_73/act/max_abs":10.9375,"train/train/layer__model_layers_44/param/norm":21.45048463358404,"train/train/tensor_act_model_layers_23_mlp_up_proj/norm":2059.606422883099,"train/train/tensor_act_model_layers_20_self_attn_k_proj/mean":-0.0107269287109375,"train/train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp_down_proj/max_abs":0.416015625,"train/train/tensor_act_model_layers_22_self_attn/norm":297.2261457100618,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/norm":0.0014408702903855955,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/max_abs":0.19140625,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/max_abs":0.00022602081298828125,"train/train/layer__model_layers_22/param/std":0.04987709453095696,"train/train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_input_layernorm/norm":5792.608032231382,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_49_mlp_gate_proj_weight/mean":-5.53131103515625e-05,"train/train/layer_model_layers_25/act/mean":0.0032862254551478793,"train/train/tensor_act_model_layers_80_mlp_gate_proj/std":0.5078125685453369,"train/train/tensor_grad_model_layers_53_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_q_proj/std":1.0703135024017318,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/norm":0.02234403623384255,"train/train/tensor_param_model_layers_37_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/norm":0.0006041539966716422,"train/train/tensor_act_model_layers_55_mlp_down_proj/max_abs":0.5390625,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/mean":1.1343217920511961e-07,"train/train/tensor_act_model_layers_61_self_attn/std":0.12317689075960862,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/norm":5.09375,"train/train/tensor_param_model_layers_66_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_post_attention_layernorm/norm":5792.614135743488,"train/train/tensor_param_model_layers_24_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/std":7.239512342991092e-05,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/mean":-1.6450881958007812e-05,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/norm":4.65625,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/std":8.324185770963107e-05,"train/train/tensor_act_model_layers_64_mlp_down_proj/max_abs":0.83203125,"train/train/layer__model_layers_90/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_mlp_gate_proj_weight/norm":7.78125,"train/train/tensor_param_model_layers_33_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/mean":-7.632188498973846e-07,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/mean":2.062879502773285e-07,"train/train/layer_model_layers_44/grad/norm":0.037838624332861906,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_mlp_gate_proj_weight/mean":-8.249282836914062e-05,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/max_abs":0.0002155303955078125,"train/train/tensor_param_model_layers_12_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp_up_proj/std":0.27148450964643034,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_56_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/std":0.046142578125,"train/train/tensor_act_model_layers_28_self_attn_q_proj/norm":5750.665140819769,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/std":7.321119818853538e-05,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/norm":0.01058383751904764,"train/train/tensor_act_model_layers_79/norm":9898.11848348886,"train/train/tensor_act_model_layers_62_self_attn_v_proj/mean":0.004425048828125,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/norm":3.453125,"train/train/tensor_act_model_layers_37_self_attn/norm":653.9716916525397,"train/train/tensor_act_model_layers_67_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/std":5.7234921154163555e-05,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/max_abs":0.1748046875,"train/train/tensor_act_model_layers_62_post_attention_layernorm/mean":0.00028133392333984375,"train/train/tensor_act_model_layers_76_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_44/param/std":0.0529334786081155,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/mean":-5.098991096019745e-08,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn/norm":1404.9770829646761,"train/train/tensor_act_model_layers_85_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer_model_layers_32/grad/norm":0.0388477683283716,"train/train/tensor_act_model_layers_21_self_attn_v_proj/norm":2030.577894272509,"train/train/tensor_act_model_layers_24_post_attention_layernorm/std":1.0000016931430074,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/mean":2.671731635928154e-07,"train/train/tensor_act_model_layers_75_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_63/act/std":0.6476507428370961,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp_down_proj/norm":208.38878066925434,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/norm":4.3125,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/mean":-0.0001735687255859375,"train/train/tensor_act_model_layers_55_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/mean":-0.0004711151123046875,"train/train/tensor_act_model_layers_3_self_attn_o_proj/max_abs":0.48046875,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/norm":0.0019021573947222042,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_q_proj/mean":0.04180908203125,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/std":2.6377171224696722e-05,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_k_proj/std":0.8906293004124534,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/norm":0.01802739235322792,"train/train/tensor_grad_model_layers_3_mlp_gate_proj_weight/mean":1.2645614333450794e-07,"train/train/tensor_act_model_layers_71_self_attn_o_proj/norm":836.6121965638475,"train_loss":11.324662740071615,"train/train/tensor_act_model_layers_17_mlp/max_abs":0.51171875,"train/train/tensor_act_model_layers_63_self_attn/max_abs":1.296875,"train/train/tensor_act_model_layers_64_mlp_gate_proj/max_abs":2.515625,"train/train/tensor_act_model_layers_49_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_o_proj/max_abs":2.265625,"train/train/tensor_act_model_layers_24_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_v_proj/norm":1422.6494921544434,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/mean":-1.712469384074211e-07,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/std":0.0400390625,"train/train/tensor_act_model_layers_13_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/max_abs":0.0003223419189453125,"train/train/tensor_grad_model_layers_29_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/norm":5792.606933595026,"train/train/tensor_act_model_layers_12_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_gate_proj/std":0.3535165585853305,"train/train/tensor_act_model_layers_48_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/mean":1.5844125300645828e-07,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/std":1.4879418031241033e-05,"train/train/tensor_act_model_layers_61_mlp_gate_proj/max_abs":2.609375,"train/train/layer_model_layers_20/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/norm":0.02908829508204207,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/max_abs":0.0004558563232421875,"train/train/layer_model_layers_44/act/std":0.6295661613177178,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/std":5.2402085413193415e-05,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/std":0.0546875,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn_k_proj/mean":0.023040771484375,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/norm":4.75,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/norm":5.53125,"train/train/tensor_act_model_layers_85_post_attention_layernorm/norm":5792.607788087887,"train/train/tensor_act_model_layers_36_self_attn_v_proj/norm":2460.3511614771937,"train/train/tensor_act_model_layers_54/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/std":0.0272216796875,"train/train/layer_model_layers_90/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn/max_abs":2.71875,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/max_abs":0.171875,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/mean":-1.380685716867447e-07,"train/train/tensor_param_model_layers_25_input_layernorm_weight/max_abs":1,"train/train/layer__model_layers_51/param/norm":21.504814141256837,"train/train/tensor_act_model_layers_46_mlp_up_proj/std":0.3300786845645803,"train/train/tensor_act_model_layers_62_self_attn/norm":550.3042435967333,"train/train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp_up_proj/mean":0.01806640625,"train/train/tensor_act_model_layers_35_self_attn/max_abs":0.6953125,"train/train/tensor_act_model_layers_52_mlp_down_proj/norm":427.2502918834834,"train/train/tensor_act_model_layers_35_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_4/param/mean":0.0015038090079511383,"train/train/tensor_act_model_layers_24_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn_q_proj/std":0.8613487228214299,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/std":6.148136637215685e-05,"train/train/tensor_act_model_layers_57_mlp/mean":0.0003428459167480469,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/norm":5.75,"train/train/tensor_act_model_layers_18_input_layernorm/norm":5792.614135742881,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/max_abs":0.15234375,"train/train/tensor_act_model_layers_64_mlp_gate_proj/std":0.4106454123488691,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp_up_proj/mean":0.0050811767578125,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/mean":-4.0745362639427185e-09,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/std":5.3662131518424354e-05,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/mean":-0.0002117156982421875,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/max_abs":0.00043487548828125,"train/train/tensor_act_model_layers_32_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_v_proj/mean":0.009429931640625,"train/train/tensor_act_model_layers_0_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/max_abs":0.0986328125,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/mean":2.076849341392517e-07,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/norm":0.017034167468567358,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/std":1.7186828668233e-05,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/norm":4.59375,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/std":4.87345250443951e-05,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/norm":6.4375,"train/train/tensor_act_model_layers_43_self_attn_k_proj/std":0.7763692112062578,"train/train/tensor_act_model_layers_86_self_attn_v_proj/mean":-0.0120697021484375,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/std":0.043212890625,"train/train/tensor_act_model_layers_85_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_85_input_layernorm/mean":0.00630950927734375,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/max_abs":0.15234375,"train/train/tensor_act_model_layers_64_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_3/param/max_abs":1,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/norm":3.703125,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/std":8.071513182291789e-05,"train/train/tensor_act_model_layers_70/max_abs":10.6875,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/max_abs":0.001190185546875,"train/train/layer_model_layers_78/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/mean":-0.0001468658447265625,"train/train/tensor_act_model_layers_51_self_attn_k_proj/mean":-0.026092529296875,"train/train/tensor_act_model_layers_36_self_attn_q_proj/mean":-0.02032470703125,"train/train/tensor_act_model_layers_88_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_88/param/norm":24.88339996363037,"train/train/tensor_act_model_layers_1_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/max_abs":0.0005950927734375,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/std":0.0274658203125,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_61/grad/std":7.746121213615061e-05,"train/train/tensor_param_model_layers_54_mlp_gate_proj_weight/max_abs":0.1484375,"train/train/tensor_act_model_layers_34_input_layernorm/max_abs":6.6875,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/mean":3.762543201446533e-06,"train/train/tensor_act_model_layers_69_self_attn_o_proj/norm":793.9336190313563,"train/train/tensor_act_model_layers_11_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/norm":0.013708440384789826,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/std":0.025634765625,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/mean":5.8673322200775146e-08,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/std":6.343288650244293e-05,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_44_self_attn_k_proj/std":0.7724628264883184,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_27/act/max_abs":8.875,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/max_abs":0.000278472900390625,"train/train/tensor_act_model_layers_29_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/mean":9.250652510672808e-08,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/std":1.9509008305328243e-05,"train/train/tensor_act_model_layers_56_self_attn_q_proj/std":0.9375041846092531,"train/train/tensor_param_model_layers_47_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_down_proj/std":0.039551189106821226,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/max_abs":0.0004367828369140625,"train/train/layer_model_layers_88/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/std":0.0400390625,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp_up_proj/max_abs":1.5859375,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/std":4.91571239964438e-05,"train/train/tensor_act_model_layers_72_mlp_down_proj/mean":0.0008287429809570312,"train/train/tensor_act_model_layers_23_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/std":8.602106296571222e-05,"train/train/tensor_act_model_layers_47_self_attn_v_proj/mean":0.0040130615234375,"train/train/tensor_param_model_layers_65_mlp_gate_proj_weight/max_abs":0.1611328125,"train/train/tensor_act_model_layers_25_self_attn_v_proj/max_abs":2.765625,"train/train/layer_model_layers_33/act/norm":13712.915903575495,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/max_abs":0.00037384033203125,"train/train/layer_model_layers_86/grad/max_abs":0.001556396484375,"train/train/tensor_act_model_layers_60_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/norm":5.25,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/norm":0.010058447457147813,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/norm":0.010820063164657957,"train/train/tensor_act_model_layers_4_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_post_attention_layernorm/norm":5792.609741213678,"train/train/tensor_param_model_layers_27_input_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_58/param/max_abs":1,"train/train/tensor_act_model_layers_30_self_attn_k_proj/max_abs":4.9375,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/std":8.612959047648706e-05,"train/train/tensor_act_model_layers_76_self_attn_k_proj/norm":5583.980210224692,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/max_abs":0.000392913818359375,"train/train/tensor_act_model_layers_30_mlp_gate_proj/max_abs":2.046875,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/std":0.03369140625,"train/train/tensor_act_model_layers_3/norm":7502.9394876826345,"train/train/tensor_act_model_layers_28_mlp_gate_proj/norm":2137.4906250911145,"train/train/tensor_act_model_layers_82_self_attn_q_proj/max_abs":6.71875,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_15/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/norm":0.022231463946111165,"train/train/tensor_act_model_layers_23_self_attn_v_proj/mean":0.0013885498046875,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/max_abs":0.232421875,"train/train/tensor_act_model_layers_22_input_layernorm/mean":-0.020721435546875,"train/train/tensor_grad_model_layers_52_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/norm":5.09375,"train/train/tensor_param_model_layers_13_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/std":4.9785147925132896e-05,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/mean":5.088746547698975e-06,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_gate_proj/max_abs":2.125,"train/train/tensor_act_model_layers_19_post_attention_layernorm/std":1.0000013168892534,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/mean":6.277114152908325e-07,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/norm":3.046875,"train/train/tensor_act_model_layers_29/mean":-0.019989013671875,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/mean":-2.2049061954021454e-07,"train/train/tensor_param_model_layers_81_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp_up_proj/norm":2655.8547060643264,"train/train/tensor_act_model_layers_64_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/mean":-9.529292583465576e-06,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/std":3.398210777858682e-05,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/std":8.466348430617687e-05,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/norm":0.0037779988294828894,"train/train/layer_model_layers_50/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn_v_proj/mean":-0.0010738372802734375,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/norm":3.234375,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/max_abs":0.0003948211669921875,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/max_abs":9.679794311523438e-05,"train/train/tensor_act_model_layers_15_input_layernorm/mean":-0.027862548828125,"train/train/layer__model_layers_39/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp/std":0.08093289445987627,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn/std":0.24267649068815048,"train/train/tensor_act_model_layers_47_mlp_up_proj/norm":2746.104741080747,"train/train/tensor_act_model_layers_80_self_attn/mean":0.000904083251953125,"train/train/tensor_act_model_layers_62_mlp_down_proj/norm":585.4367700284329,"train/train/tensor_act_model_layers_25_mlp_down_proj/norm":217.99589836417505,"train/train/tensor_act_model_layers_33_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp_gate_proj/norm":2920.6243845802405,"train/train/layer_model_layers_24/grad/mean":-1.8993314128956072e-08,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/norm":0.004269789814427087,"train/train/tensor_act_model_layers_30_self_attn/mean":-0.0002536773681640625,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_mlp_down_proj/mean":0.001071929931640625,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/std":8.371291298982635e-05,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/norm":0.03027675979596435,"train/train/tensor_act_model_layers_5_post_attention_layernorm/mean":-0.03106689453125,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp_down_proj/std":0.03631607668455286,"train/train/tensor_act_model_layers_39_self_attn_k_proj/norm":5211.996649948288,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/norm":0.0026663833406494154,"train/train/layer_model_layers_39/act/max_abs":9.4375,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/max_abs":0.23046875,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_18/param/norm":20.126388975748355,"train/train/tensor_act_model_layers_88_self_attn_q_proj/norm":6284.805042975773,"train/train/layer__model_layers_12/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp_gate_proj/mean":0.002857208251953125,"train/train/tensor_param_model_layers_32_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/norm":6.71875,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/std":1.6679012909159717e-05,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/max_abs":0.255859375,"train/train/tensor_act_model_layers_8_mlp_up_proj/std":0.23144533085923583,"train/train/layer__model_layers_54/param/std":0.055336287917294454,"train/train/tensor_act_model_layers_88_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn_o_proj/std":0.1818866492320301,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/norm":5.4375,"train/train/tensor_act_model_layers_77_mlp_gate_proj/max_abs":2.703125,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/max_abs":0.00113677978515625,"train/train/tensor_act_model_layers_1_mlp/norm":1323.0838183677952,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/std":9.315515510514776e-05,"train/train/tensor_act_model_layers_63/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44/std":1.2421880206033526,"train/train/tensor_act_model_layers_13_self_attn_o_proj/norm":328.41662974550667,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_71/grad/norm":0.054936857240720775,"train/train/tensor_grad_model_layers_22_mlp_gate_proj_weight/norm":0.014090233846308394,"train/train/tensor_act_model_layers_45_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/norm":4.3125,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/max_abs":0.1640625,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/max_abs":0.0004100799560546875,"train/train/tensor_param_model_layers_37_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/std":0.05322265625,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn/max_abs":0.9921875,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_down_proj/norm":1066.3355648733557,"train/train/tensor_act_model_layers_3_mlp_up_proj/std":0.27539064698185395,"train/train/tensor_param_model_layers_51_mlp_gate_proj_weight/mean":-0.0003490447998046875,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/std":2.6760316565109004e-05,"train/train/tensor_act_model_layers_49_self_attn/max_abs":1.84375,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/max_abs":0.1416015625,"train/train/tensor_act_model_layers_49/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp_down_proj/norm":1016.727593024412,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/norm":0.017000564614905313,"train/train/tensor_act_model_layers_86_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/std":0.032958984375,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_67/grad/norm":0.058077330422788606,"train/train/tensor_act_model_layers_44/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3/max_abs":8,"train/train/tensor_act_model_layers_85_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_q_proj/norm":5390.874220577439,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/std":0.043212890625,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/std":0.030029296875,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/max_abs":0.00067901611328125,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/norm":10.5,"train/train/tensor_act_model_layers_92_self_attn_v_proj/max_abs":3.015625,"train/train/tensor_act_model_layers_70_mlp_gate_proj/max_abs":2.53125,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/mean":0.0002899169921875,"train/train/tensor_act_model_layers_37_self_attn_v_proj/mean":0.002292633056640625,"train/train/tensor_act_model_layers_62_mlp_gate_proj/mean":0.00543975830078125,"train/train/tensor_act_model_layers_25_mlp_gate_proj/mean":-0.001918792724609375,"train/train/tensor_act_model_layers_35_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_v_proj/mean":-0.0040283203125,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_58/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_input_layernorm/max_abs":5.5,"train/train/tensor_act_model_layers_80_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/max_abs":0.3125,"train/train/tensor_param_model_layers_52_mlp_gate_proj_weight/max_abs":0.14453125,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/max_abs":0.162109375,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/norm":0.005544927214767395,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/norm":0.02232860892474167,"train/train/tensor_act_model_layers_12_mlp/std":0.04242028985560395,"train/train/tensor_act_model_layers_43_self_attn_o_proj/std":0.11133511007597366,"train/train/tensor_act_model_layers_12_self_attn_k_proj/mean":-0.03839111328125,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/max_abs":8.296966552734375e-05,"train/train/tensor_act_model_layers_78/mean":0.0035648345947265625,"train/train/tensor_act_model_layers_38_self_attn_k_proj/norm":4886.365385506599,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/norm":3.359375,"train/train/tensor_act_model_layers_3_self_attn/norm":161.02770829767573,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/max_abs":0.00021266937255859375,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/std":3.787938887246659e-05,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/std":0.0263671875,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/max_abs":0.193359375,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/norm":0.0005768996578341603,"train/train/tensor_act_model_layers_18_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/max_abs":0.1455078125,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/mean":-0.00017833709716796875,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_input_layernorm/norm":5792.610717773946,"train/train/tensor_act_model_layers_91_self_attn_v_proj/max_abs":3.078125,"train/train/tensor_grad_model_layers_37_mlp_gate_proj_weight/std":4.9670101327467654e-05,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp/norm":253.3229871528563,"train/train/layer__model_layers_78/param/mean":0.0016011016417219188,"train/train/tensor_act_model_layers_86_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_o_proj/norm":459.95871946283614,"train/train/tensor_act_model_layers_69_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/mean":-3.38914105668664e-08,"train/train/tensor_act_model_layers_73/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_16/act/norm":14493.40980268287,"train/train/layer_model_layers_74/grad/max_abs":0.0013580322265625,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/max_abs":0.000522613525390625,"train/train/tensor_act_model_layers_60_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/norm":3.59375,"train/train/tensor_act_model_layers_43_self_attn/max_abs":1.5546875,"train/train/tensor_act_model_layers_71_post_attention_layernorm/std":1.00000151406015,"train/train/tensor_param_model_layers_7_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21/std":1.2675846993109814,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/mean":-0.00018215179443359375,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_40/act/norm":13699.217454086216,"train/train/tensor_act_model_layers_7_self_attn_v_proj/max_abs":1.9375,"train/train/tensor_act_model_layers_11_self_attn_o_proj/std":0.052430626464179446,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/std":0.0556640625,"train/train/layer_model_layers_61/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_q_proj/norm":5799.51555987015,"train/train/tensor_act_model_layers_12_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn/mean":0.00234222412109375,"train/train/tensor_act_model_layers_93_self_attn_o_proj/norm":1312.0046012674302,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/mean":1.1082738637924194e-07,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/max_abs":0.0004024505615234375,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/std":0.037353515625,"train/train/tensor_act_model_layers_12_mlp_gate_proj/mean":-5.7637691497802734e-05,"train/train/tensor_act_model_layers_23_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp_down_proj/mean":-0.0004715919494628906,"train/train/layer__model_layers_52/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp/norm":259.868483762349,"train/train/tensor_act_model_layers_82_mlp_up_proj/norm":4319.2325461621485,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_gate_proj_weight/norm":0.014961209605957084,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/mean":-0.00019931793212890625,"train/train/tensor_act_model_layers_23_self_attn_q_proj/norm":5210.825111142873,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/std":4.616286742331758e-05,"train/train/tensor_act_model_layers_29_input_layernorm/norm":5792.606201175934,"train/train/tensor_grad_model_layers_11_mlp_gate_proj_weight/mean":1.0339499567635357e-07,"train/train/layer__model_layers_75/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/max_abs":0.1279296875,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/norm":0.01920280865376866,"train/train/tensor_act_model_layers_27_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_v_proj/mean":-0.003849029541015625,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/std":2.5897571229996484e-05,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/max_abs":0.00048828125,"train/train/tensor_param_model_layers_62_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/grad/std":6.553575145657187e-05,"train/train/tensor_act_model_layers_8_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/max_abs":0.00019741058349609375,"train/train/layer_model_layers_34/grad/norm":0.03820659627036161,"train/train/tensor_act_model_layers_85_mlp_up_proj/mean":-0.0009160041809082031,"train/train/tensor_param_model_layers_21_mlp_gate_proj_weight/std":0.0262451171875,"train/train/tensor_act_model_layers_13_self_attn_o_proj/std":0.05676313035664301,"train/train/tensor_param_model_embed_tokens_weight/std":0.11865234375,"train/train/tensor_act_model_layers_33_self_attn/std":0.08496094085697624,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/max_abs":0.1328125,"train/train/layer_model_layers_45/act/mean":0.0032006672450474332,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/mean":1.1622905731201172e-05,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/mean":-4.693865776062012e-06,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_14/grad/mean":-1.2580555240375204e-07,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/mean":3.805325832217932e-07,"train/train/tensor_act_model_layers_41_self_attn/std":0.08838657418931434,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/std":3.922643569371218e-05,"train/train/tensor_param_model_layers_43_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/norm":0.018805564565138332,"train/train/tensor_grad_model_layers_7_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/max_abs":1.046875,"train/train/layer__model_layers_31/param/std":0.05140087179430976,"train/train/layer__model_layers_16/param/max_abs":1,"train/train/tensor_param_model_layers_6_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/max_abs":0.259765625,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/mean":-2.0442530512809753e-07,"train/train/tensor_grad_model_layers_22_mlp_gate_proj_weight/max_abs":0.0003643035888671875,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/std":0.0478515625,"train/train/layer_model_layers_23/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/max_abs":0.00054168701171875,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/mean":1.8715858459472656e-05,"train/train/tensor_act_model_layers_15_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_29/grad/norm":0.04007257032204487,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/mean":1.4842953532934189e-08,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/norm":4.375,"train/train/tensor_act_model_layers_45_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_q_proj/mean":-0.0003070831298828125,"train/train/tensor_act_model_layers_90_self_attn/std":0.23974798373178566,"train/train/tensor_param_model_layers_23_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_39/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/std":0.031494140625,"train/train/tensor_act_model_layers_25_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_51/act/mean":-0.00434131281716483,"train/train/tensor_act_model_layers_2/norm":7502.565269088194,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/mean":-1.4901161193847656e-08,"train/train/tensor_act_model_layers_79_self_attn_q_proj/mean":0.00736236572265625,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/max_abs":0.2392578125,"train/train/tensor_act_model_layers_90_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/std":0.0252685546875,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_72/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/std":0.02392578125,"train/train/tensor_param_model_layers_74_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_71_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_31_mlp_up_proj/max_abs":1.7578125,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/norm":0.018441709587448285,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_mlp_gate_proj/std":0.30029426541474924,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp/std":0.21020563520226931,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/norm":0.015233499550729898,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/norm":4.8125,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/norm":0.01662110132560154,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/std":0.051513671875,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_self_attn_v_proj/norm":2352.358332291732,"train/train/tensor_param_model_layers_25_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/max_abs":0.00021076202392578125,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/max_abs":0.000751495361328125,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/max_abs":0.0005950927734375,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/mean":3.0025839805603027e-06,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_71/grad/mean":8.041774287424668e-08,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/max_abs":0.17578125,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/max_abs":0.0015106201171875,"train/train/layer_model_layers_8/act/max_abs":8.875,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_k_proj/norm":5268.704915370326,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/mean":-0.000415802001953125,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_mlp_gate_proj/max_abs":2.734375,"train/train/tensor_act_model_layers_66_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_post_attention_layernorm/mean":-0.004138946533203125,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/std":2.1504310976049595e-05,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/norm":0.02314802054970506,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/std":0.0537109375,"train/train/tensor_act_model_layers_47_self_attn_q_proj/norm":5430.215328048495,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/mean":-4.400499165058136e-08,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_76/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/mean":-3.6496203392744064e-07,"train/train/tensor_act_model_layers_5_self_attn/std":0.06884864042635273,"train/train/tensor_act_model_layers_12_post_attention_layernorm/max_abs":5.78125,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2/mean":-0.025146484375,"train/train/tensor_act_model_layers_69_mlp_gate_proj/std":0.44335947792959485,"train/train/tensor_act_model_layers_43_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_v_proj/norm":2176.556892776109,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_75_self_attn/mean":-0.0022735595703125,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/norm":4.125,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/std":0.03466796875,"train/train/tensor_param_model_layers_60_mlp_gate_proj_weight/mean":9.965896606445312e-05,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/norm":0.011674080590512533,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/norm":0.024588685193996714,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/std":7.432372184888043e-05,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/mean":8.856877684593201e-07,"train/train/tensor_act_model_layers_64_mlp/std":0.10681206039204338,"train/train/tensor_act_model_layers_54_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_input_layernorm/std":1.0000001639127598,"train/train/tensor_param_model_layers_59_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/std":0.0289306640625,"train/train/tensor_act_model_layers_68_mlp_down_proj/max_abs":1.0859375,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_mlp_gate_proj_weight/max_abs":0.00115966796875,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/max_abs":0.11767578125,"train/train/tensor_act_model_layers_86_self_attn/max_abs":3.1875,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/norm":0.005779375887295546,"train/train/tensor_act_model_layers_9_mlp_up_proj/norm":1856.9382844472304,"train/train/tensor_grad_model_layers_32_mlp_gate_proj_weight/max_abs":0.000469207763671875,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/norm":0.025702541309906398,"train/train/layer_model_layers_79/grad/std":7.715100631962532e-05,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/norm":0.01655437270077161,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_71_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_2/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/max_abs":0.130859375,"train/train/tensor_act_model_layers_83_mlp_gate_proj/norm":4382.706671189889,"train/train/tensor_act_model_layers_84_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_33/grad/max_abs":0.00156402587890625,"train/train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/mean":-4.7046050895005465e-06,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/max_abs":0.12890625,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/norm":3.5,"train/train/layer_model_layers_55/grad/frac_near_user_limit":0,"train/train/layer_model_layers_14/grad/norm":0.04052822295385669,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/mean":-5.473848432302475e-07,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/mean":0.0001773834228515625,"train/train/tensor_act_model_layers_60_post_attention_layernorm/mean":0.00035762786865234375,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/std":5.7068062298993434e-05,"train/train/layer__model_layers_93/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/max_abs":0.00019931793212890625,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/max_abs":0.16015625,"train/train/tensor_act_model_layers_85_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_k_proj/mean":-0.0318603515625,"train/train/tensor_act_model_layers_39_mlp_gate_proj/norm":2533.302216871165,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/norm":8.0625,"train/train/tensor_param_model_layers_27_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_post_attention_layernorm/mean":-0.0140228271484375,"train/train/tensor_act_model_layers_53_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_v_proj/std":0.44482724025686254,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/std":5.769459976602411e-05,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/mean":2.948567271232605e-06,"train/train/tensor_param_model_layers_13_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_mlp_gate_proj_weight/std":0.036865234375,"train/train/tensor_act_model_layers_50_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48/std":1.2578130341278446,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/mean":1.1593103408813477e-05,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/std":0.0311279296875,"train/train/tensor_act_model_layers_36_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/mean":-7.976777851581573e-06,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/norm":0.013516496839683322,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/std":2.9935052106989974e-05,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/max_abs":1.96875,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/std":8.132613693036825e-05,"train/train/tensor_act_model_layers_15_mlp/norm":230.43217809229014,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/norm":0.008176333523144218,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/max_abs":0.10888671875,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/max_abs":0.1396484375,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/norm":4.03125,"train/train/layer_model_layers_51/grad/norm":0.040331136838472344,"train/train/tensor_act_model_layers_70_self_attn_q_proj/max_abs":5.9375,"train/train/tensor_act_model_layers_12_mlp_down_proj/norm":245.78391475750408,"train/train/tensor_act_model_layers_36_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/max_abs":0.0004177093505859375,"train/train/tensor_act_model_layers_53_mlp/mean":0.0008840560913085938,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/max_abs":0.00018978118896484375,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/max_abs":0.002471923828125,"train/train/tensor_param_model_layers_27_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_input_layernorm/max_abs":6.1875,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/norm":0.013784151904908849,"train/train/tensor_act_model_layers_51_post_attention_layernorm/std":1.0000004967440548,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/norm":0.006423533374012426,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_mlp_gate_proj_weight/max_abs":0.0009918212890625,"train/train/tensor_act_model_layers_9_mlp_up_proj/std":0.22656261294691996,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/max_abs":0.0001964569091796875,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_36/param/max_abs":1,"train/train/tensor_act_model_layers_55_self_attn_q_proj/norm":4847.008767158688,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/max_abs":0.00012302398681640625,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/norm":0.0257307164244128,"train/train/layer__model_layers_10/param/max_abs":1,"train/train/layer_model_layers_63/grad/std":6.455979586014796e-05,"train/train/tensor_param_model_layers_65_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_q_proj/norm":6316.694301052238,"train/train/tensor_param_model_layers_26_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_65/grad/mean":1.9121025914149054e-07,"train/train/tensor_grad_model_layers_80_mlp_gate_proj_weight/norm":0.02899536109686309,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/std":3.310240115654972e-05,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/std":0.033203125,"train/train/tensor_act_model_layers_13_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_v_proj/std":0.4218750911054218,"train/train/tensor_act_model_layers_92_self_attn_k_proj/mean":0.07421875,"train/train/layer_model_layers_15/act/norm":13673.391866596387,"train/train/tensor_act_model_layers_31_self_attn_v_proj/mean":0.0017261505126953125,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/std":4.0857550673731656e-05,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/std":0.0498046875,"train/train/tensor_act_model_layers_58_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_mlp_gate_proj_weight/max_abs":0.000579833984375,"train/train/layer_model_layers_38/grad/mean":6.866869451289244e-09,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_89/param/max_abs":1,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/std":0.039306640625,"train/train/tensor_param_model_layers_22_mlp_gate_proj_weight/std":0.0267333984375,"train/train/tensor_param_model_layers_5_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_42_post_attention_layernorm/mean":-0.014129638671875,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/mean":-1.0700896382331848e-06,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/mean":4.0279701352119446e-07,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/norm":0.0019546145481184637,"train/train/tensor_grad_model_layers_52_mlp_gate_proj_weight/max_abs":0.00069427490234375,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/max_abs":0.00145721435546875,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/max_abs":0.00013256072998046875,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/norm":0.015578678132761996,"train/train/tensor_act_model_layers_40_mlp_down_proj/mean":0.0002522468566894531,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/std":4.9814590228697956e-05,"train/train/tensor_act_model_layers_42_mlp_up_proj/mean":-0.002197265625,"train/train/tensor_param_model_layers_18_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_12/act/norm":13795.339736375314,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/std":0.061279296875,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/mean":0.0001068115234375,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/norm":0.016688288630734856,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/mean":2.9685907065868378e-08,"train/train/tensor_act_model_layers_63_self_attn/norm":474.0419062117446,"train/train/layer__model_layers_39/param/max_abs":1,"train/train/tensor_act_model_layers_15_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/norm":0.0019432503703231856,"train/train/tensor_param_model_layers_82_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/std":0.041015625,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/max_abs":0.000698089599609375,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/max_abs":0.0004558563232421875,"train/train/tensor_act_model_layers_22_mlp_gate_proj/norm":2037.031153603405,"train/train/tensor_act_model_layers_64_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_77/grad/max_abs":0.00152587890625,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/std":0.0302734375,"train/train/tensor_grad_model_layers_47_mlp_gate_proj_weight/norm":0.018684136422882004,"train/train/layer_model_layers_50/act/norm":13777.703825059805,"train/train/tensor_act_model_layers_70/std":1.4668037392951143,"train/train/tensor_act_model_layers_77_self_attn_v_proj/mean":-0.00567626953125,"train/train/tensor_param_model_layers_14_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_q_proj/mean":0.04071044921875,"train/train/tensor_act_model_layers_57_self_attn_o_proj/norm":759.7304074187987,"train/train/tensor_act_model_layers_87_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/std":0.032958984375,"train/train/tensor_act_model_layers_82/mean":0.00677490234375,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_19/act/max_abs":8.4375,"train/train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_24/act/norm":13311.157530702534,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/std":0.050537109375,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/norm":0.003729962531789594,"train/train/tensor_act_model_layers_0_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_2/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/mean":0.0002574920654296875,"train/train/layer__model_layers_49/param/max_abs":1,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/std":0.0001129192140619457,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/mean":9.269570000469685e-09,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/norm":10.0625,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_7/grad/max_abs":0.001312255859375,"train/train/layer__model_layers_84/param/mean":0.0013248663796649523,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/std":0.00010661051370849723,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/mean":2.5529880076646805e-07,"train/train/tensor_act_model_layers_66_input_layernorm/mean":0.003177642822265625,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_q_proj/norm":5412.189935222259,"train/train/tensor_act_model_layers_91_self_attn_o_proj/mean":0.0044097900390625,"train/train/tensor_act_model_layers_38_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57/frac_near_user_limit":0,"train/train/layer__model_layers_87/param/mean":0.0016443309099551482,"train/train/tensor_act_model_layers_57_self_attn_k_proj/std":0.8906253576277969,"train/train/tensor_act_model_layers_82_self_attn_v_proj/norm":2702.4328455824498,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/std":0.0238037109375,"train/train/tensor_act_model_layers_34_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/norm":0.0007497457284966477,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/mean":0.0003986358642578125,"train/train/layer_model_layers_91/grad/norm":0.07339265982895919,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_post_attention_layernorm/max_abs":6.4375,"train/train/tensor_param_model_layers_60_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn_k_proj/norm":4493.494438284077,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/std":0.0001508622942463791,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/grad/norm":0.04280286255770404,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/max_abs":0.0002536773681640625,"train/train/tensor_act_model_layers_38_mlp_down_proj/std":0.05541992828143223,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/mean":-0.00018596649169921875,"train/train/tensor_act_model_layers_63_post_attention_layernorm/mean":0.0020182132720947266,"train/train/tensor_act_model_layers_76/std":1.6074288297745503,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/std":4.280402076932651e-05,"train/train/tensor_act_model_layers_74_self_attn_o_proj/norm":1198.9102010576203,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/mean":8.200004231184721e-09,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp_down_proj/mean":-0.0003800392150878906,"train/train/tensor_act_model_layers_10_self_attn_k_proj/norm":5484.77787510298,"train/train/tensor_act_model_layers_92_mlp_up_proj/max_abs":4.375,"train/train/tensor_act_model_layers_53_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_post_attention_layernorm/max_abs":6.3125,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/mean":-3.546476364135742e-06,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/mean":1.6450881958007812e-05,"train/train/tensor_act_model_layers_23/norm":7324.08072965263,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/std":2.2033683309205222e-05,"train/train/layer_model_layers_11/grad/mean":-1.6495742357093346e-08,"train/train/tensor_act_model_layers_57_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_q_proj/std":0.9033252839647561,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/max_abs":0.0002536773681640625,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/mean":-5.1280949264764786e-08,"train/train/tensor_act_model_layers_91_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81/mean":0.003692626953125,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/mean":7.171183824539185e-08,"train/train/tensor_act_model_layers_84_mlp_gate_proj/std":0.54394800843976,"train/train/tensor_act_model_layers_90_mlp_down_proj/mean":0.00630950927734375,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/std":5.126815646622825e-05,"train/train/tensor_grad_model_layers_23_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18/max_abs":8.4375,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/norm":0.0003985057965407417,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/norm":0.0012961160247864983,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp_down_proj/max_abs":1.09375,"train/train/tensor_act_model_layers_47_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/std":7.558930034400605e-05,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/mean":-9.417999535799026e-08,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/mean":-2.4563632905483246e-08,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/max_abs":0.0001201629638671875,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/std":0.0311279296875,"train/train/tensor_param_model_layers_55_input_layernorm_weight/max_abs":1,"train/train/layer__model_layers_85/param/mean":0.001699444656253047,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_49/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93/norm":19570.269254857278,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/max_abs":0.11376953125,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/mean":-9.275972843170166e-07,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/norm":0.028841355655271465,"train/train/tensor_act_model_layers_18_mlp_up_proj/max_abs":2.0625,"train/train/layer_model_layers_32/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_68_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/norm":0.0220213456582989,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/mean":-0.00018787384033203125,"train/train/tensor_param_model_layers_31_mlp_gate_proj_weight/norm":5.0625,"train/train/tensor_act_model_layers_13_input_layernorm/norm":5792.605102540415,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/mean":-0.000591278076171875,"train/train/tensor_act_model_layers_71_mlp_up_proj/norm":3701.7925167991775,"train/train/tensor_act_model_layers_70_self_attn_k_proj/norm":5159.270432075472,"train/train/tensor_act_model_layers_7_mlp_gate_proj/max_abs":2.046875,"train/train/layer_model_layers_28/act/mean":0.0013737678527832031,"train/train/tensor_param_model_layers_26_mlp_gate_proj_weight/norm":4.875,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/mean":1.0826624929904938e-08,"train/train/tensor_act_model_layers_92_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/std":2.2584251585218488e-05,"train/train/tensor_act_model_layers_2_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn/norm":283.9690646290212,"train/train/tensor_act_model_layers_12_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_22/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/mean":-4.102475941181183e-07,"train/train/tensor_act_model_layers_80_self_attn/max_abs":3.578125,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/max_abs":0.0006561279296875,"train/train/tensor_act_model_layers_67_self_attn_o_proj/std":0.12219601847799517,"train/train/tensor_act_model_layers_47_mlp/norm":378.067815680335,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_post_attention_layernorm/max_abs":5.78125,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/std":0.02685546875,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/std":4.481550922238992e-05,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/std":6.36318289941326e-05,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_o_proj/norm":810.8501244792732,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/max_abs":0.00019931793212890625,"train/train/tensor_act_model_layers_84_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_mlp_gate_proj_weight/mean":-1.6196281649172306e-08,"train/train/tensor_act_model_layers_90_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/norm":0.026706281173012946,"train/train/tensor_act_model_layers_74_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_mlp_gate_proj_weight/mean":8.440017700195312e-05,"train/train/layer__model_layers_64/param/mean":0.001588818435549922,"train/train/tensor_act_model_layers_16/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_mlp_gate_proj_weight/mean":-4.103640094399452e-09,"train/train/layer__model_layers_45/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/max_abs":0.0004634857177734375,"train/train/tensor_act_model_layers_31_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/mean":4.7907233238220215e-06,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/mean":2.0372681319713593e-08,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/norm":0.001174524163963411,"train/train/tensor_act_model_layers_92/mean":0.03564453125,"train/train/tensor_grad_model_layers_49_mlp_gate_proj_weight/norm":0.01908293223133037,"train/train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/norm":0.01439889709641589,"train/train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_10/grad/std":5.222081057648656e-05,"train/train/tensor_act_model_layers_36/norm":7192.857926801638,"train/train/tensor_param_model_layers_52_input_layernorm_weight/std":0,"train/train/layer__model_layers_71/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_mlp_gate_proj/max_abs":1.8984375,"train/train/layer_model_layers_22/act/max_abs":8.5625,"train/train/tensor_act_model_layers_27_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_mlp_gate_proj_weight/mean":-4.6253204345703125e-05,"train/train/tensor_act_model_layers_6_self_attn_o_proj/max_abs":0.65234375,"train/train/layer_model_layers_70/act/mean":0.0022444043840680805,"train/train/tensor_act_model_layers_86_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp/norm":350.24356742412095,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/max_abs":0.1748046875,"train/train/tensor_act_model_layers_46/max_abs":9.4375,"train/train/tensor_param_model_layers_57_mlp_gate_proj_weight/mean":-0.000446319580078125,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/mean":-3.647804260253906e-05,"train/train/tensor_param_model_layers_63_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/norm":3.875,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/max_abs":0.0003070831298828125,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_58/act/max_abs":9.8125,"train/train/tensor_act_model_layers_24_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/mean":4.763714969158173e-06,"train/train/tensor_param_model_layers_11_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_93/act/std":1.2419200997729598,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/norm":0.003014257820310823,"train/train/tensor_act_model_layers_46_self_attn_v_proj/max_abs":2.609375,"train/train/tensor_act_model_layers_14_post_attention_layernorm/max_abs":5.96875,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/mean":3.9080623537302017e-07,"train/train/layer_model_layers_29/act/std":0.6334281585576697,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_lm_head/std":2.0937609316412384,"train/train/tensor_param_model_layers_63_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/std":1.2493973449065554e-05,"train/train/tensor_act_model/mean":0.0063323974609375,"train/train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/std":3.76293772919897e-05,"train/train/tensor_act_model_layers_86_post_attention_layernorm/std":1.0000008296625507,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/max_abs":0.000629425048828125,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/mean":3.183959051966667e-08,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/norm":7.75,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/mean":3.688001015689224e-10,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/mean":1.5966594219207764e-05,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/mean":-3.337860107421875e-05,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/norm":5.125,"train/train/tensor_act_model_layers_29_self_attn_v_proj/std":0.35546896132189065,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/norm":0.01574023786072125,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/norm":0.0021268776099931373,"train/train/tensor_grad_model_layers_13_mlp_gate_proj_weight/norm":0.015761967199700203,"train/train/tensor_act_model_layers_62_self_attn_v_proj/max_abs":2.640625,"train/train/tensor_act_model_layers_28_mlp_up_proj/mean":-0.0014801025390625,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/norm":4.09375,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/mean":5.817413330078125e-05,"train/train/tensor_grad_model_layers_32_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/std":7.022774868359502e-05,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/std":7.511756959383399e-05,"train/train/tensor_param_model_layers_62_mlp_gate_proj_weight/norm":6.53125,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_30/act/max_abs":9.0625,"train/train/tensor_grad_model_layers_20_mlp_gate_proj_weight/std":4.073161675922301e-05,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/mean":-0.0004673004150390625,"train/train/tensor_param_model_layers_65_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_69/norm":8429.674688395582,"train/train/tensor_act_model_layers_78_self_attn/mean":-0.0006682872772216797,"train/train/tensor_act_model_layers_42_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_q_proj/norm":5459.367488086289,"train/train/layer__model_layers_56/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_k_proj/norm":4992.328862329741,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/mean":0.00063323974609375,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_27/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_norm_weight/max_abs":0.006378173828125,"train/train/tensor_act_model_layers_93_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/norm":5.6875,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/max_abs":0.1318359375,"_step":106,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/max_abs":0.000164031982421875,"train/train/tensor_grad_model_layers_25_mlp_gate_proj_weight/norm":0.014365288885390475,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/mean":-2.0890729501843452e-07,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_gate_proj_weight/norm":4.25,"train/train/tensor_act_model_layers_83_self_attn_o_proj/std":0.23734831686790042,"train/train/tensor_act_model_layers_72_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_post_attention_layernorm/max_abs":6.5625,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/max_abs":0.2490234375,"train/train/tensor_act_model_layers_21_mlp_gate_proj/norm":2047.6994403468857,"train/train/tensor_act_model_layers_55_self_attn_o_proj/max_abs":0.9921875,"train/train/tensor_act_model_layers_77_self_attn/norm":1636.79448506998,"train/train/tensor_act_model_layers_47_mlp/std":0.06542974325900593,"train/train/tensor_act_model_layers_2_input_layernorm/mean":-0.021881103515625,"train/train/tensor_act_model_layers_58_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/norm":0.018012300871172594,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/max_abs":0.2451171875,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_input_layernorm/mean":0.0143280029296875,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_down_proj/mean":-0.00081634521484375,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/std":0.052490234375,"train/train/layer_model_layers_46/act/norm":13922.560611317054,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/std":0.052001953125,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/max_abs":0.25,"train/train/layer_model_layers_21/act/norm":13970.461655984043,"train/train/layer_model_layers_24/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21/max_abs":8.5,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/mean":-0.000209808349609375,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/max_abs":0.00018310546875,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/std":2.708514698159075e-05,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/mean":1.437729224562645e-08,"train/train/layer__model_layers_43/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_k_proj/mean":-0.02850341796875,"train/train/tensor_act_model_layers_91_self_attn_q_proj/std":1.3203125687745882,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/norm":0.005582736138324168,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/mean":-6.87548890709877e-07,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/max_abs":0.0004291534423828125,"train/train/tensor_act_model_layers_7_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/mean":-5.676411092281342e-07,"train/train/layer__model_layers_17/param/max_abs":1,"train/train/tensor_grad_model_layers_27_mlp_gate_proj_weight/mean":1.21421180665493e-07,"train/train/tensor_param_model_layers_20_mlp_gate_proj_weight/std":0.0264892578125,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/norm":4.375,"train/train/tensor_act_model_layers_84_mlp_down_proj/norm":1365.1432541772751,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/mean":-5.566980689764023e-07,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/norm":0.0009045438759999216,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/std":0.029052734375,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/std":8.148050135925842e-05,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/mean":-8.833594620227814e-07,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_mlp_gate_proj/max_abs":2.03125,"train/train/tensor_act_model_layers_37_post_attention_layernorm/max_abs":6.6875,"train/train/layer__model_layers_92/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/norm":7.09375,"train/train/tensor_act_model_layers_18_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/norm":5.0625,"train/train/layer_model_layers_26/grad/norm":0.04099847274925563,"train/train/layer_model_layers_18/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_40/grad/mean":-1.7645729592363473e-08,"train/train/layer__model_layers_64/param/norm":22.840650439512444,"train/train/tensor_act_model_layers_31_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/mean":-4.6253204345703125e-05,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_22/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_q_proj/norm":3991.070923029374,"train/train/tensor_act_model_layers_83_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_o_proj/std":0.07007038840059213,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/mean":-1.8190621631219983e-07,"train/train/layer_model_layers_75/act/mean":-0.004104682377406529,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/std":0.039794921875,"train/train/tensor_act_model_layers_77_self_attn_q_proj/norm":6815.787754404414,"train/train/tensor_grad_model_layers_66_mlp_gate_proj_weight/std":7.714198742230954e-05,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_gate_proj_weight/std":7.9405321336589e-05,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/norm":0.0014934673771555673,"train/train/tensor_act_model_layers_73_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp_gate_proj/max_abs":1.75,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/max_abs":0.2470703125,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/norm":0.015063972403157265,"train/train/tensor_act_model_layers_49_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/norm":7.90625,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/mean":0.00013065338134765625,"train/train/tensor_act_model_layers_42_mlp_gate_proj/norm":2629.418802599971,"train/train/tensor_param_model_layers_87_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_73_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn_k_proj/norm":4271.696577025091,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn_o_proj/norm":485.8187210742259,"train/train/tensor_act_model_layers_5_input_layernorm/mean":-0.03253173828125,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/norm":0.014434298290355252,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_12/grad/std":4.81907825696277e-05,"train/train/tensor_act_model_layers_40_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp/std":0.04101672365709403,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/max_abs":0.12890625,"train/train/tensor_act_model_layers_17_mlp_down_proj/std":0.043701519241331026,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/std":1.9615333586657996e-05,"train/train/tensor_act_model_layers_26_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp/std":0.0737304807585093,"train/train/tensor_act_model_layers_20_self_attn_q_proj/max_abs":5.03125,"train/train/tensor_param_model_layers_67_mlp_gate_proj_weight/mean":2.9921531677246094e-05,"train/train/tensor_param_model_layers_54_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/max_abs":0.130859375,"train/train/layer_model_layers_5/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/norm":6.5625,"train/train/tensor_act_model_layers_45/norm":7204.9904905911,"train/train/tensor_act_model_layers_76_mlp_gate_proj/norm":3924.1991319868334,"train/train/layer_model_layers_92/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_post_attention_layernorm/norm":5792.60327148757,"train/train/tensor_act_model_layers_6_mlp_down_proj/std":0.07031589295263843,"train/train/tensor_act_model_layers_44_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_v_proj/mean":-0.001506805419921875,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/mean":5.163019523024559e-08,"train/train/tensor_act_model_layers_48_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn/std":0.03949226088889081,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/std":0.00010123678837725502,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/norm":6.625,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/norm":0.022315508193567868,"train/train/tensor_act_model_layers_27_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_post_attention_layernorm/mean":-0.0318603515625,"train/train/tensor_act_model_layers_47_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/std":0.00011435992864951747,"train/train/tensor_act_model_layers_62_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_mlp_gate_proj_weight/norm":6.90625,"train/train/tensor_param_model_layers_44_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_78_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/mean":4.1996827349066734e-08,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/std":0.0322265625,"train/train/tensor_act_model_layers_68_self_attn_v_proj/norm":2806.7872238213363,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/mean":-4.315376281738281e-05,"train/train/tensor_grad_model_layers_15_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/norm":0.021996040616507315,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/norm":0.01449379354250784,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/mean":0.0002002716064453125,"train/train/tensor_act_model_layers_74_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_63_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_21/param/std":0.05055470460205569,"train/train/layer_model_layers_14/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/norm":0.001256848680510162,"train/train/tensor_act_model_layers_85_mlp_up_proj/frac_near_user_limit":0,"train/train/layer_model_layers_25/act/std":0.6301522380926997,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/mean":0.0003814697265625,"train/train/tensor_act_model_layers_79_self_attn_v_proj/max_abs":2.9375,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/norm":0.009669113091455152,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/std":3.9221213007869856e-05,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/std":3.5659519211000994e-05,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/std":3.609736684443521e-05,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/max_abs":0.0005035400390625,"train/train/tensor_act_model_layers_28_self_attn_q_proj/max_abs":7.5625,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/mean":2.852175384759903e-07,"train/train/layer_model_layers_26/grad/max_abs":0.0013885498046875,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/norm":0.025872499644598045,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/norm":0.0031500900637196363,"train/train/tensor_param_model_norm_weight/mean":1,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_mlp_gate_proj_weight/std":5.879087538526016e-05,"train/train/tensor_act_model_layers_51_self_attn/max_abs":1.2734375,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/norm":0.016184562460754108,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/mean":5.8673322200775146e-08,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/std":0.00023006676381077303,"train/train/tensor_act_model_layers_40_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_mlp_gate_proj_weight/norm":0.023384511175374705,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/max_abs":0.000263214111328125,"train/train/tensor_act_model_layers_61_self_attn_v_proj/mean":-0.002391815185546875,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_post_attention_layernorm/norm":5792.604614267046,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/std":0.0262451171875,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_58_self_attn_v_proj/max_abs":2.890625,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_19_self_attn_o_proj/mean":-0.00019156932830810547,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/std":0.025634765625,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_o_proj/std":0.18750207811763583,"train/train/tensor_grad_model_layers_65_mlp_gate_proj_weight/norm":0.027408930883353083,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/mean":4.000961780548096e-06,"train/train/tensor_act_model_layers_9_input_layernorm/max_abs":5.34375,"train/train/tensor_grad_model_layers_41_mlp_gate_proj_weight/norm":0.017798316227376928,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_65_mlp_down_proj/mean":0.0014019012451171875,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_post_attention_layernorm/norm":5792.612182622017,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/max_abs":0.12890625,"train/train/tensor_act_model_layers_68_self_attn_q_proj/mean":-0.05584716796875,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/std":6.254481332869265e-05,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/mean":2.261251211166382e-06,"train/train/tensor_act_model_layers_66_self_attn_k_proj/max_abs":6.5,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/mean":7.715076208114624e-06,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/max_abs":0.002685546875,"train/train/tensor_act_model_layers_87_mlp_down_proj/max_abs":2.28125,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/norm":0.01841047358993128,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/std":0.025146484375,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/norm":0.014322078564716432,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/max_abs":0.1767578125,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/norm":0.005960603330902186,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/std":9.875772421910525e-05,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/max_abs":0.000530242919921875,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/max_abs":0.24609375,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/mean":2.300366759300232e-07,"train/train/tensor_act_model_layers_23_self_attn_k_proj/std":0.7265625158625263,"train/train/layer_model_layers_89/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_v_proj/std":0.4379891110628192,"train/train/tensor_act_model_layers_38_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/max_abs":0.1279296875,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/max_abs":0.000270843505859375,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/max_abs":0.000682830810546875,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/max_abs":0.0003337860107421875,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/norm":0.005361852498632587,"train/train/layer_model_layers_83/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/max_abs":0.00018310546875,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/norm":0.012977185206254033,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_norm/std":1.000000642728785,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/max_abs":0.00023174285888671875,"train/train/layer__model_layers_13/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_q_proj/mean":-0.021026611328125,"train/train/layer__model_layers_53/param/std":0.054107107724641966,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/std":3.000771949787032e-05,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/norm":0.006159752215075044,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/mean":0.000263214111328125,"train/train/tensor_act_model_layers_23_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49/max_abs":9.8125,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/mean":-2.753734588623047e-05,"train/train/tensor_grad_model_layers_41_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_v_proj/std":0.36572378314982257,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn/mean":7.62939453125e-05,"train/train/tensor_act_model_layers_44_self_attn_v_proj/norm":2079.7327548189082,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/norm":5.90625,"train/train/tensor_grad_model_layers_85_mlp_gate_proj_weight/norm":0.029361145729478413,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_embed_tokens_weight/mean":-3.427267074584961e-06,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/mean":9.489059448242188e-05,"train/train/tensor_act_model_layers_67_self_attn_o_proj/norm":706.9723338155231,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/norm":0.015664408577689453,"train/train/tensor_act_model_layers_84_mlp_gate_proj/mean":0.002033233642578125,"train/train/tensor_act_model_layers_59_mlp_down_proj/std":0.10009766675350086,"train/train/tensor_act_model_layers_43_self_attn_q_proj/max_abs":6,"train/train/tensor_act_model_layers_35/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/mean":1.0932853911072016e-07,"train/train/tensor_act_model_layers_79/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn_q_proj/max_abs":5.875,"train/train/tensor_param_model_layers_60_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_gate_proj_weight/mean":-2.7894973754882812e-05,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/max_abs":0.216796875,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/std":0.00011437834035348584,"train/train/layer_model_layers_46/act/std":0.6423790501409666,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/max_abs":0.11962890625,"train/train/tensor_act_model_layers_42_post_attention_layernorm/norm":5792.603393561679,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/mean":4.76837158203125e-05,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp/mean":0.0005617141723632812,"train/train/tensor_act_model_layers_65_self_attn_o_proj/norm":954.5216318918541,"train/train/global/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_v_proj/norm":2117.5539109364154,"train/train/tensor_act_model_layers_15_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp_up_proj/std":0.24438528374869245,"train/train/tensor_act_model_layers_45/mean":-0.0140228271484375,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/mean":-0.00017547607421875,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/norm":0.003398850309029703,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_v_proj/norm":2430.148407499675,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/max_abs":0.00037384033203125,"train/train/tensor_act_model_layers_88_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/std":0.04052734375,"train/train/tensor_act_model_layers_47_mlp_down_proj/max_abs":0.53515625,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_gate_proj_weight/norm":0.022259597631752552,"train/train/tensor_act_model_layers_68_mlp_up_proj/max_abs":2.5625,"train/train/tensor_act_model_layers_11_self_attn_k_proj/mean":0.029266357421875,"train/train/tensor_act_model_layers_22_mlp_gate_proj/max_abs":1.828125,"train/train/tensor_act_model_layers_39_self_attn_k_proj/std":0.8994157254745235,"train/train/tensor_act_model_layers_33_input_layernorm/std":1.0000007211926587,"train/train/tensor_act_model_layers_66_self_attn/std":0.1997145803489047,"train/train/tensor_act_model_layers_45_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn_o_proj/std":0.1129172474534032,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/std":0.0001361985664258936,"train/train/tensor_grad_model_layers_30_mlp_gate_proj_weight/std":4.298962341471513e-05,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/norm":7,"train/train/tensor_act_model_layers_9_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_q_proj/std":1.1562508453385383,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp_down_proj/norm":599.8585135776389,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/norm":4.03125,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_q_proj/max_abs":6.5625,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/mean":-0.00020503997802734375,"train/train/tensor_act_model_layers_33_mlp/mean":0.0005474090576171875,"train/train/layer_model_layers_75/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/norm":0.002513419471345485,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/mean":9.611248970031738e-07,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_gate_proj/mean":8.678436279296875e-05,"train/train/tensor_act_model_layers_58_self_attn/mean":0.0001899339258670807,"train/train/tensor_param_model_layers_55_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_64_self_attn_q_proj/std":0.9062500657706402,"train/train/tensor_param_model_layers_93_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_q_proj/mean":-0.06689453125,"train/train/tensor_act_model_layers_52/max_abs":9.9375,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/mean":1.1265277862548828e-05,"train/train/tensor_act_model_layers_68_mlp_gate_proj/mean":-0.0113983154296875,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/mean":0.000133514404296875,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_o_proj/max_abs":0.76171875,"train/train/tensor_act_model_layers_75_self_attn_q_proj/mean":-0.07470703125,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/norm":0.01825216522828329,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/mean":3.361701965332031e-05,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/norm":0.01664044930808578,"train/train/tensor_act_model_layers_25_self_attn_o_proj/max_abs":1.0546875,"train/train/layer_model_layers_90/grad/norm":0.07026421069229546,"train/train/tensor_grad_model_layers_71_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/max_abs":0.0002498626708984375,"train/train/tensor_act_model_layers_4/std":1.2949265527424967,"train/train/tensor_act_model_layers_40_self_attn_q_proj/mean":-0.001834869384765625,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/std":8.046073691600968e-05,"train/train/tensor_act_model_layers_3_self_attn/max_abs":0.48046875,"train/train/tensor_act_model_layers_82_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_mlp_gate_proj_weight/max_abs":0.00054931640625,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/mean":-3.886222839355469e-05,"train/train/tensor_act_model_layers_93_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/norm":5.46875,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/std":0.04833984375,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/norm":0.01507271803132069,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/mean":-2.0954757928848267e-07,"train/train/tensor_act_model_layers_33_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_post_attention_layernorm/max_abs":6.8125,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/max_abs":0.11865234375,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_18/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/norm":0.0016615754610824762,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/mean":1.1874362826347351e-07,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/max_abs":0.000537872314453125,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/norm":3.375,"train/train/tensor_grad_model_layers_47_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_post_attention_layernorm/std":1.0000019835695975,"train/train/tensor_act_model_layers_66_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/mean":-0.000213623046875,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/norm":4.53125,"train/train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn/mean":-0.0005130767822265625,"train/train/layer__model_layers_50/param/norm":21.81861594953092,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/std":0.04931640625,"train/train/tensor_act_model_layers_82_self_attn_k_proj/max_abs":6.28125,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/norm":0.0151221605248327,"train/train/tensor_act_model_layers_38_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp/max_abs":0.57421875,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/max_abs":0.000270843505859375,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/std":3.677531713808683e-05,"train/train/tensor_act_model_layers_12_mlp/mean":-0.0001284480094909668,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_35/grad/mean":7.172927330139088e-08,"train/train/layer__model_layers_81/param/std":0.060719819456443065,"train/train/tensor_act_model_layers_64_input_layernorm/std":1.0000005056583492,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_42/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_mlp_down_proj/norm":378.067815680335,"train/train/tensor_act_model_layers_77_input_layernorm/mean":0.00637054443359375,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/norm":9.375,"train/train/tensor_act_model_layers_30_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn_v_proj/norm":2330.397847000113,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_rotary_emb/norm":2297.209228515625,"train/train/tensor_grad_model_norm_weight/norm":0.0683811697957076,"train/train/tensor_act_model_layers_24_mlp_down_proj/std":0.036865237286153894,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/mean":-9.604264050722122e-08,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/max_abs":0.000301361083984375,"train/train/tensor_act_model_layers_16_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn/mean":0.0004253387451171875,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/std":1.610681423594592e-05,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/mean":6.246566772460938e-05,"train/train/layer_model_layers_83/grad/max_abs":0.00170135498046875,"train/train/tensor_act_model_layers_88_input_layernorm/max_abs":5.65625,"train/train/tensor_act_model_layers_0_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/mean":5.3783878684043884e-08,"train/train/layer_model_layers_55/act/std":0.6341862642347129,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/std":1.286117129170658e-05,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/mean":-1.4650868251919746e-07,"train/train/tensor_act_model_layers_34/std":1.2402401055744272,"train/train/tensor_act_model_layers_38_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_post_attention_layernorm/norm":5792.612304690319,"train/train/layer_model_layers_23/grad/max_abs":0.00131988525390625,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_k_proj/mean":0.0372314453125,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/mean":-0.00014209747314453125,"train/train/tensor_grad_model_layers_34_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_k_proj/std":0.8144555827945336,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/mean":-7.772445678710938e-05,"train/train/layer_model_layers_4/grad/max_abs":0.00179290771484375,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/mean":0.000324249267578125,"train/train/tensor_param_model_layers_64_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_gate_proj_weight/norm":0.02924163697173194,"train/train/tensor_param_model_layers_92_mlp_gate_proj_weight/norm":10.125,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_13/grad/mean":-4.9177391596963734e-08,"train/train/tensor_act_model_layers_61_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/mean":4.3427280616015196e-07,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/max_abs":0.1318359375,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/mean":1.3843178749084473e-05,"train/train/layer_model_layers_38/grad/norm":0.04152853104644706,"train/train/tensor_param_model_layers_73_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer_model_layers_91/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_21_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_mlp_gate_proj_weight/max_abs":0.166015625,"train/train/tensor_act_model_layers_79/mean":0.00247955322265625,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp_gate_proj/norm":2753.944523125706,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/std":5.3004270068682556e-05,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_mlp_gate_proj_weight/max_abs":0.0004634857177734375,"train/train/tensor_param_model_layers_46_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/max_abs":0.00016117095947265625,"train/train/tensor_act_model_layers_92_self_attn_o_proj/max_abs":2.640625,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_57_mlp_down_proj/mean":0.0003428459167480469,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/max_abs":0.1083984375,"train/train/tensor_act_model_layers_39_mlp_gate_proj/std":0.3090832560790135,"train/train/tensor_grad_model_layers_53_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/std":8.138905492430643e-05,"train/train/tensor_act_model_layers_42_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn_q_proj/norm":4991.394051927548,"train/train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_54/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_mlp_gate_proj/mean":-0.0069732666015625,"train/train/tensor_param_model_layers_62_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_mlp_gate_proj_weight/max_abs":0.1220703125,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/max_abs":0.283203125,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/norm":0.02551028591624583,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_42/param/norm":21.579935206474207,"train/train/tensor_act_model_layers_19_mlp_down_proj/mean":0.0011043548583984375,"train/train/tensor_act_model_layers_36_mlp/mean":0.0003476142883300781,"train/train/tensor_act_model_layers_74_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_5/grad/norm":0.04915100831460435,"train/train/tensor_grad_model_layers_93_mlp_gate_proj_weight/std":0.00010962134278903208,"train/train/tensor_act_model_layers_20_self_attn/std":0.03820977964088249,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/max_abs":0.0013885498046875,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/max_abs":0.00051116943359375,"train/train/layer_model_layers_15/grad/mean":9.962826874252415e-08,"train/train/tensor_act_model_layers_78_post_attention_layernorm/std":1.0000013414237499,"train/train/tensor_act_model_layers_65_self_attn_o_proj/std":0.16504068831668958,"train/train/tensor_act_model_layers_70_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/max_abs":0.000713348388671875,"train/train/tensor_act_model_layers_73_mlp_gate_proj/std":0.4570313201755486,"train/train/layer_model_layers_57/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/norm":0.0240217347820979,"train/train/tensor_act_model_layers_35_self_attn_o_proj/mean":0.00015103816986083984,"train/train/tensor_param_model_layers_51_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_61_mlp_gate_proj/mean":-0.0007123947143554688,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/std":0.03759765625,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/norm":0.001071359611646758,"train/train/layer_model_layers_9/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_mlp_down_proj/max_abs":0.89453125,"train/train/tensor_act_model_layers_0_input_layernorm/std":1.0000000162890499,"train/train/tensor_act_model_layers_63_input_layernorm/std":1.000000508154115,"train/train/tensor_act_model_layers_32/std":1.24805258360081,"train/train/layer_model_layers_85/act/std":0.8382457260837101,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_gate_proj_weight/mean":9.1552734375e-05,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/mean":-0.00010776519775390625,"train/train/tensor_act_model_layers_93_input_layernorm/std":1.0000009894133306,"train/train/tensor_act_model_layers_30_post_attention_layernorm/norm":5792.607788098361,"train/train/tensor_act_model_layers_16_self_attn/mean":-0.0004475116729736328,"train/train/tensor_act_model_layers_46_self_attn/std":0.09607083543679609,"train/train/tensor_param_model_layers_35_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/mean":6.876885890960693e-06,"train/train/tensor_act_model_layers_66/max_abs":10.5,"train/train/tensor_act_model_layers_69_self_attn/mean":-0.0004181861877441406,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/max_abs":0.27734375,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/max_abs":0.000881195068359375,"train/train/tensor_act_model_layers_22_self_attn_q_proj/norm":5736.857645058753,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer__model_layers_39/param/std":0.05213515307689496,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/norm":0.0072027788048007,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/std":7.94442449756177e-05,"train/train/tensor_act_model_layers_9_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_mlp_gate_proj_weight/norm":0.02515078017198333,"train/train/tensor_act_model_layers_67/norm":8192.643193963113,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/std":0.032470703125,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/max_abs":0.0003070831298828125,"train/train/tensor_act_model_layers_92_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_mlp_down_proj/norm":765.7445531230093,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/std":7.791260420136885e-05,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/std":0.0263671875,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/max_abs":0.001617431640625,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/std":0.039306640625,"train/train/tensor_act_model_layers_26_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_o_proj/max_abs":2.140625,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/max_abs":0.0007171630859375,"train/train/tensor_act_model_layers_79_mlp_down_proj/norm":1050.8473044482769,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/norm":0.014470674839885437,"train/train/layer_model_layers_48/act/max_abs":9.75,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/norm":7.84375,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/max_abs":0.25390625,"train/train/layer_model_layers_6/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/norm":7.6875,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_v_proj/std":0.5146514750658835,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/norm":0.016255473810546133,"train/train/tensor_act_model_layers_27_self_attn/mean":0.000804901123046875,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/mean":-3.257300704717636e-07,"train/train/tensor_act_model_layers_89_self_attn_q_proj/norm":5945.377223719812,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/mean":0.0002727508544921875,"train/train/tensor_param_model_layers_28_mlp_gate_proj_weight/std":0.0274658203125,"train/train/tensor_act_model_layers_86_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_63/grad/mean":7.7731291329061e-08,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/norm":0.0066016441543314515,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/norm":0.0040609597991119765,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/norm":0.00069216785756226,"train/train/tensor_act_model_layers_22_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/norm":0.0014420867537308243,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/std":0.051513671875,"train/train/layer_model_layers_75/act/std":0.7395020165262619,"train/train/layer_model_layers_20/act/max_abs":8.5,"train/train/tensor_act_model_layers_16_post_attention_layernorm/mean":-0.02862548828125,"train/train/tensor_param_model_layers_50_mlp_gate_proj_weight/mean":0.000148773193359375,"train/train/layer__model_layers_12/param/max_abs":1,"train/train/tensor_act_model_layers_1_self_attn_o_proj/norm":93.98442972853606,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/max_abs":0.000858306884765625,"train/train/tensor_param_model_layers_28_mlp_gate_proj_weight/max_abs":0.11181640625,"train/train/layer__model_layers_15/param/norm":20.30618291193904,"train/train/tensor_act_model_layers_6_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/mean":-0.000152587890625,"train/train/tensor_act_model_layers_37_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92/norm":16371.063102941976,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/max_abs":0.00019741058349609375,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/mean":0.000274658203125,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/std":0.038330078125,"train/train/tensor_act_model_layers_46_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/std":7.75018736606698e-05,"train/train/tensor_param_model_layers_47_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/norm":0.031207831064638047,"train/train/tensor_param_model_layers_9_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_69/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/norm":4.40625,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_mlp_up_proj/mean":-0.001949310302734375,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/mean":-4.454050213098526e-07,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/norm":0.002759584801615805,"train/train/tensor_act_model_layers_32_self_attn_k_proj/mean":0.0088653564453125,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/max_abs":0.1337890625,"train/train/tensor_param_model_layers_69_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_29_self_attn_q_proj/std":0.9179691010332451,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/max_abs":0.150390625,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn/max_abs":1.5,"train/train/layer__model_layers_32/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/norm":0.0010496952448951632,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/norm":0.007790175967659126,"train/train/layer__model_layers_62/param/norm":22.356989084791806,"train/train/tensor_grad_model_norm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/mean":3.871973603963852e-07,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_input_layernorm/mean":-0.0104827880859375,"train/train/tensor_grad_model_layers_87_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_mlp_gate_proj_weight/max_abs":0.00067138671875,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_gate_proj_weight/std":4.906617688484679e-05,"train/train/tensor_act_model_layers_65_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/std":5.260413598139955e-05,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/norm":0.017695906547023372,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/max_abs":0.0005340576171875,"train/train/layer_model_layers_3/act/norm":13957.08110488496,"train/train/tensor_act_model_layers_76_post_attention_layernorm/norm":5792.605346680185,"train/train/tensor_act_model_layers_68_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_input_layernorm/mean":0.0007314682006835938,"train/train/tensor_act_model_layers_50_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/std":0.0267333984375,"train/train/tensor_param_model_layers_24_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_15_self_attn/std":0.060791918687008686,"train/train/layer_model_layers_82/act/mean":0.005508286612374442,"train/train/global/act/mean":-0.05404644128860976,"train/train/tensor_param_model_layers_42_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/std":0.03466796875,"train/train/tensor_act_model_layers_58_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn_k_proj/mean":0.10009765625,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_post_attention_layernorm/max_abs":5.53125,"train/train/tensor_act_model_layers_87_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/max_abs":0.236328125,"train/train/tensor_act_model_layers_49/mean":-0.0108642578125,"train/train/layer_model_layers_85/act/max_abs":11.75,"train/train/tensor_act_model_layers_16_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_82/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/mean":0.0002574920654296875,"train/train/layer_model_layers_74/act/mean":0.003414954457964216,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/std":0.04931640625,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/std":0.04345703125,"train/train/tensor_act_model_layers_63_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/mean":-0.0002346038818359375,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/std":0.034423828125,"train/train/tensor_act_model_layers_55_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/mean":-5.7334545999765396e-08,"train/train/tensor_act_model_layers_13_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/norm":0.01577333728783012,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_post_attention_layernorm/max_abs":4.875,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/norm":0.016755345480842475,"train/train/tensor_act_model_layers_62_self_attn_k_proj/std":0.8437519872330734,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/max_abs":0.00152587890625,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/mean":1.3216049410402775e-07,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/norm":0.004320568628117232,"train/train/layer_model_layers_49/act/norm":13611.137575485696,"train/train/layer_model_layers_38/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_input_layernorm/norm":5792.612915040456,"train/train/tensor_act_model_layers_40_self_attn/std":0.0922858085957331,"train/train/layer_model_layers_8/grad/max_abs":0.0013580322265625,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/max_abs":0.00156402587890625,"train/train/tensor_act_model_layers_8/norm":7475.456310970054,"train/train/tensor_act_model_layers_27_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_67/param/norm":23.07079796517667,"train/train/tensor_act_model_layers_72_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_74/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_v_proj/mean":-0.0043792724609375,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/mean":-7.772905519232154e-08,"train/train/layer_model_layers_1/act/max_abs":8.3125,"train/train/tensor_grad_model_layers_17_mlp_gate_proj_weight/mean":6.012851372361183e-08,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_93/param/max_abs":1,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/std":7.502538659995334e-05,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/max_abs":0.234375,"train/train/layer_model_layers_4/act/std":0.7093115376069966,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/norm":0.024393836226080018,"train/train/tensor_act_model_layers_49_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33/std":1.2402400790105899,"train/train/tensor_act_model_layers_63_self_attn_q_proj/mean":0.048583984375,"train/train/tensor_act_model_layers_9_mlp/std":0.03936784714423077,"train/train/tensor_act_model_layers_41_self_attn/norm":512.1543658271897,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_12/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_v_proj/std":0.3300781459464911,"train/train/tensor_grad_model_layers_18_mlp_gate_proj_weight/max_abs":0.0006866455078125,"train/train/tensor_act_model_layers_8_input_layernorm/mean":-0.032012939453125,"train/train/tensor_act_model_layers_12_self_attn/max_abs":0.91796875,"train/train/layer__model_layers_27/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/norm":5.96875,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_25/param/norm":20.73900781890252,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/mean":0.0004711151123046875,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/std":8.793278994286089e-05,"train/train/layer_model_layers_10/grad/norm":0.04230223866881385,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_gate_proj/std":0.356934643082725,"train/train/layer_model_layers_10/act/std":0.6364571369701842,"train/train/tensor_act_model_layers_18_input_layernorm/std":1.0000012456431677,"train/train/tensor_act_model_layers_89_self_attn_v_proj/mean":0.00354766845703125,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_o_proj/std":0.12317689075960862,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/norm":0.019505006324798336,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/max_abs":0.1787109375,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/max_abs":0.2109375,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/norm":7.625,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/norm":0.0015163457618526722,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/norm":6.5,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_mlp_gate_proj_weight/norm":9,"train/train/tensor_act_model_layers_59_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/norm":0.02127939757296583,"train/train/tensor_act_model_layers_25_mlp_up_proj/mean":-0.0011844635009765625,"train/train/tensor_param_model_layers_36_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_71/param/mean":0.00152292117685684,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/max_abs":0.00045013427734375,"train/train/tensor_param_model_layers_40_mlp_gate_proj_weight/std":0.0306396484375,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/std":0.00017042465450034457,"train/train/tensor_act_model_layers_37_mlp_down_proj/std":0.05322266180375818,"train/train/tensor_act_model_layers_63_self_attn_v_proj/norm":2117.9370692195093,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_26/act/mean":-0.003111072949000767,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/std":0.00011721916668372076,"train/train/tensor_act_model_layers_31/frac_near_dtype_limit":0,"train/train/layer_model_layers_59/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/norm":0.0006873524523420133,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/max_abs":0.0016937255859375,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/std":0.0546875,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/std":7.368747288810797e-05,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/max_abs":0.00011968612670898438,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/norm":5.96875,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp_up_proj/norm":5433.6237823436295,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/std":0.034912109375,"train/train/layer_model_layers_5/grad/max_abs":0.00144195556640625,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/mean":1.500447979196906e-07,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/std":0.049072265625,"train/train/layer_model_layers_23/grad/std":3.8476667972101557e-05,"train/train/tensor_act_model_layers_37_input_layernorm/max_abs":6.8125,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/std":2.1659894383637416e-05,"train/train/tensor_act_model_layers_72_input_layernorm/std":1.000001349914781,"train/train/tensor_act_model_layers_35_self_attn_o_proj/std":0.03949226088889081,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/norm":0.019848535216069253,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_mlp_gate_proj_weight/mean":0.00018024444580078125,"train/train/tensor_act_model_layers_0/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_90_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/max_abs":0.1279296875,"train/train/tensor_act_model_layers_70_self_attn_k_proj/max_abs":4.875,"train/train/layer__model_layers_85/param/frac_near_user_limit":0,"train/train/layer_model_layers_9/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12/mean":-0.03387451171875,"train/train/tensor_act_model_norm/norm":5792.6140136753575,"train/train/tensor_act_model_layers_74_mlp_up_proj/std":0.46289074446077055,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/norm":5.09375,"train/train/layer_model_layers_17/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/mean":9.202957153320312e-05,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn/norm":759.7304074187987,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/max_abs":0.00127410888671875,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_post_attention_layernorm/max_abs":5.71875,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/mean":2.4767359718680382e-08,"train/train/tensor_grad_model_layers_68_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/max_abs":0.25,"train/train/layer__model_layers_86/param/norm":25.44435181061801,"train/train/tensor_act_model_layers_50_mlp_up_proj/max_abs":2.15625,"train/train/tensor_grad_model_layers_53_mlp_gate_proj_weight/std":5.6168944989707764e-05,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/mean":-1.2386590242385864e-07,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/std":0.0228271484375,"train/train/tensor_act_model_layers_6_self_attn/std":0.06408932496959693,"train/train/tensor_act_model_layers_69_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_3/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/norm":4.1875,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/norm":0.21388335145109846,"train/train/layer_model_layers_4/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_36/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_mlp_up_proj/std":0.3715834419574836,"train/train/tensor_act_model_layers_59_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/mean":-8.940696716308594e-06,"train/train/tensor_act_model_layers_66_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_10_mlp_up_proj/max_abs":2.328125,"train/train/tensor_act_model_layers_14_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_71_post_attention_layernorm/norm":5792.605957032412,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/std":3.777029992990393e-05,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/std":0.0262451171875,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/max_abs":0.1630859375,"train/train/tensor_act_model_layers_51_mlp/mean":0.0007076263427734375,"train/train/tensor_act_model_layers_62_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/mean":-0.000339508056640625,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/std":4.329201181281416e-05,"train/train/layer__model_layers_54/param/norm":22.424156025088212,"train/train/tensor_act_model_rotary_emb/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn_v_proj/mean":0.00036597251892089844,"train/train/layer__model_layers_90/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp/norm":222.48812308687425,"train/train/tensor_act_model_layers_24_mlp_down_proj/max_abs":0.34765625,"train/train/layer_model_layers_87/act/norm":19078.08243947046,"train/train/layer__model_layers_0/param/max_abs":1,"train/train/layer__model_layers_66/param/max_abs":1,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/norm":0.015321891006937522,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/mean":0.000141143798828125,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/max_abs":0.0001678466796875,"train/train/tensor_grad_model_layers_21_mlp_gate_proj_weight/mean":-7.182825356721878e-08,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/mean":-6.118789315223694e-07,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/max_abs":0.00014400482177734375,"train/train/tensor_act_model_layers_60/max_abs":9.875,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/max_abs":0.208984375,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/mean":-7.677078247070312e-05,"train/train/tensor_act_model_layers_18_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/std":7.380223873993395e-05,"train/train/layer_model_layers_11/act/max_abs":8.625,"train/train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_input_layernorm/norm":5792.603027347218,"train/train/tensor_act_model_layers_46/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/std":3.6714980021926005e-05,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/std":0.03369140625,"train/train/tensor_act_model_layers_1_post_attention_layernorm/max_abs":4.96875,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/std":1.3276832028829799e-05,"train/train/tensor_act_model_layers_13_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/max_abs":0.000568389892578125,"train/train/layer_model_layers_62/grad/norm":0.050526264485211476,"train/train/tensor_act_model_layers_27_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp_up_proj/max_abs":2.0625,"train/train/tensor_act_model_layers_47_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_mlp_gate_proj_weight/max_abs":0.1640625,"train/train/tensor_act_model_layers_72_self_attn_v_proj/mean":-3.123283386230469e-05,"train/train/tensor_act_model_layers_5_mlp_gate_proj/std":0.22705131688705268,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/max_abs":0.00159454345703125,"train/train/layer_model_layers_9/grad/mean":-2.029651191881778e-07,"train/train/layer_model_layers_87/grad/mean":-1.299943246075218e-07,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/std":9.243633418108463e-05,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/max_abs":0.00102996826171875,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/mean":-7.771886885166168e-07,"train/train/tensor_act_model_layers_66_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/std":0.03173828125,"train/train/tensor_act_model_layers_72_self_attn_q_proj/std":0.9960939519545406,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/max_abs":0.0015869140625,"train/train/tensor_act_model_layers_20_self_attn_o_proj/mean":-0.00018978118896484375,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/mean":-1.919688656926155e-07,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/mean":-1.2695789337158203e-05,"train/train/tensor_act_model_layers_39_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_74/grad/norm":0.06294822921154188,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/mean":-3.2997922971844673e-07,"train/train/tensor_act_model_layers_56_self_attn_q_proj/norm":5428.920643120489,"train/train/tensor_act_model_layers_82_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_o_proj/max_abs":0.9296875,"train/train/tensor_act_model_layers_72_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/std":0.037841796875,"train/train/tensor_param_model_layers_35_mlp_gate_proj_weight/max_abs":0.1328125,"train/train/tensor_act_model_layers_14_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/std":0.05712890625,"train/train/layer_model_layers_49/act/mean":0.0015275137765066965,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/norm":0.0008503909120112378,"train/train/tensor_act_model_layers_50_mlp_gate_proj/mean":-0.00199127197265625,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/std":5.0985822243784325e-05,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/mean":3.528594970703125e-05,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/max_abs":0.1923828125,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/max_abs":0.1220703125,"train/train/tensor_act_model_layers_44_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/mean":2.9765069484710693e-06,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/std":1.8904250820210183e-05,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/max_abs":0.2265625,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/mean":-0.00017261505126953125,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/std":5.5695973606172484e-05,"train/train/tensor_act_model_layers_76/norm":9312.099413236607,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/max_abs":0.2138671875,"train/train/tensor_act_model_layers_39/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/mean":-9.010545909404755e-08,"train/train/tensor_act_model_layers_85_self_attn_v_proj/mean":-0.0048828125,"train/train/tensor_grad_model_layers_93_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_up_proj/std":0.2314453228113518,"train/train/tensor_act_model_layers_71_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/max_abs":0.125,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/norm":5.5625,"train/train/tensor_act_model_layers_69_mlp_up_proj/mean":0.009063720703125,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/mean":-0.000347137451171875,"train/train/layer_model_layers_45/grad/max_abs":0.0011444091796875,"train/train/tensor_param_model_layers_84_input_layernorm_weight/mean":1,"train/train/layer__model_layers_52/param/mean":0.0015460801384936257,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/max_abs":0.0003070831298828125,"train/train/tensor_param_model_layers_29_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/frac_near_dtype_limit":0,"_runtime":3382,"train/train/tensor_act_model_layers_40_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/norm":4.90625,"train/train/layer_model_layers_75/act/max_abs":10.75,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/mean":-0.0002651214599609375,"train/train/tensor_act_model_layers_68_mlp/std":0.13208078087401162,"train/train/tensor_act_model_layers_74_mlp_down_proj/max_abs":1.15625,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/std":5.0276364505005285e-05,"train/train/tensor_act_model_layers_29_mlp_down_proj/std":0.04125977833011162,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/std":0,"train/train/layer__model_layers_57/param/norm":22.447812741445436,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_mlp_gate_proj_weight/std":8.39458217114529e-05,"train/train/tensor_act_model_layers_69_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer__model_layers_20/param/mean":0.0015337225427493662,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_62/act/std":0.6589605829885319,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/max_abs":0.000385284423828125,"train/train/tensor_param_model_layers_56_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_52_mlp/max_abs":0.51953125,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_32/act/std":0.6294999081213837,"train/train/tensor_act_model_layers_9_self_attn/max_abs":0.9296875,"train/train/tensor_act_model_layers_24_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_81/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/std":0.038818359375,"train/train/tensor_act_model_layers_92_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/max_abs":0.0003814697265625,"train/train/tensor_act_model_layers_26_mlp_up_proj/mean":0.0009784698486328125,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/mean":9.1552734375e-05,"train/train/tensor_act_model_layers_43_mlp/mean":-5.019456148147583e-05,"train/train/tensor_act_model_layers_41_self_attn_v_proj/mean":-0.00033211708068847656,"train/train/tensor_act_model_layers_89_self_attn_q_proj/std":1.0234376965588097,"train/train/tensor_act_model_layers_32_self_attn_o_proj/std":0.08667385721905517,"train/train/tensor_act_model_layers_34_self_attn_k_proj/mean":0.0124664306640625,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/mean":5.456386134028435e-07,"train/train/tensor_act_model_layers_79_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp/std":0.07373049749147254,"train/train/tensor_act_model_layers_25_self_attn_o_proj/std":0.07947030556452747,"train/train/tensor_act_model_layers_59_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn_o_proj/mean":-0.0008687973022460938,"train/train/tensor_act_model_layers_29_self_attn_k_proj/mean":-0.060302734375,"train/train/tensor_act_model_layers_17_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13/norm":7429.681927653521,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/max_abs":0.1669921875,"train/train/tensor_act_model_layers_40_post_attention_layernorm/mean":-0.013671875,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn_k_proj/std":0.9873062948785484,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/max_abs":0.1396484375,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/mean":0.000431060791015625,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/mean":0.000274658203125,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_mlp_up_proj/max_abs":2.640625,"train/train/tensor_act_model_layers_71_self_attn_k_proj/norm":4769.459174335358,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/norm":8.1875,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/max_abs":0.000446319580078125,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/mean":-5.629844963550568e-07,"train/train/tensor_act_model_layers_34_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_7_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_o_proj/mean":0.0003604888916015625,"train/train/layer__model_layers_74/param/frac_near_user_limit":0,"train/train/layer__model_layers_16/param/norm":20.274086273337943,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/std":0.05419921875,"train/train/tensor_act_model_layers_77/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_mlp_gate_proj_weight/norm":0.014043326021830602,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/mean":-4.0605664253234863e-07,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/norm":6.5625,"train/train/tensor_param_model_layers_59_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_14_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/max_abs":0.11279296875,"train/train/tensor_param_model_layers_3_mlp_gate_proj_weight/norm":4.21875,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_39_mlp/std":0.0564576209557112,"train/train/tensor_act_model_layers_9_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/norm":0.0058251613862168065,"train/train/layer_model_layers_82/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_mlp_up_proj/std":0.5273438232743854,"train/train/tensor_act_model_layers_30_input_layernorm/norm":5792.607543950132,"train/train/tensor_act_model_layers_18_post_attention_layernorm/max_abs":6.21875,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_4/param/norm":19.49532320719831,"train/train/layer__model_layers_21/param/max_abs":1,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/max_abs":0.201171875,"train/train/tensor_act_model_layers_15_self_attn_o_proj/mean":-0.0006361007690429688,"train/train/tensor_act_model_layers_88_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_89/grad/max_abs":0.00131988525390625,"train/train/tensor_act_model_layers_50_mlp_down_proj/max_abs":0.51953125,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/max_abs":0.1533203125,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/std":2.1808477791243414e-05,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/std":3.523452779136642e-05,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/mean":0.00021076202392578125,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/std":5.8303513402517785e-05,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/mean":-0.0001735687255859375,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/std":5.536301596112373e-05,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/std":0.00016610180142872242,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp/norm":836.1896726311006,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/norm":0.0010697760612481861,"train/train/layer_model_layers_2/grad/mean":5.0858058349204695e-08,"train/train/tensor_act_model_layers_73_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/norm":6.90625,"train/train/tensor_act_model_layers_15_mlp_gate_proj/max_abs":2.015625,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_post_attention_layernorm/std":1.0000003166496252,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/std":6.773537063693641e-05,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/norm":0.019541356333021946,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/mean":1.1555850505828857e-05,"train/train/tensor_act_model_layers_62_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/norm":0.0015802656803287247,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/max_abs":8.535385131835938e-05,"train/train/tensor_act_model_layers_47_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/mean":3.409385681152344e-05,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/std":0.037109375,"train/train/layer__model_layers_16/param/mean":0.0015462556978841655,"train/train/tensor_param_model_layers_30_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_30_post_attention_layernorm/mean":-0.0144195556640625,"train/train/layer_model_layers_58/act/norm":14276.444327200894,"train/train/tensor_act_model_layers_16_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/max_abs":0.00109100341796875,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/max_abs":0.00019359588623046875,"train/train/tensor_grad_model_layers_30_mlp_gate_proj_weight/norm":0.01553809315691659,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10/std":1.2871143774389786,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/mean":-7.654307410120964e-08,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/norm":0.015297984327512731,"train/train/tensor_act_model_layers_9_self_attn_v_proj/mean":0.002288818359375,"train/train/tensor_act_model_layers_46_self_attn_o_proj/max_abs":1.5859375,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/std":5.2044320133805524e-05,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/mean":2.5510787963867188e-05,"train/train/tensor_param_model_layers_89_input_layernorm_weight/mean":1,"train/train/layer__model_layers_35/param/mean":0.001438182527301092,"train/train/tensor_act_model_layers_72_post_attention_layernorm/max_abs":5.53125,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90/std":2.3554810180471106,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_mlp_down_proj/max_abs":0.5859375,"train/train/layer_model_layers_44/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/std":0.06677458564038419,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/std":0.026611328125,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/max_abs":0.0002288818359375,"train/train/tensor_act_model_layers_49_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/max_abs":0.00020122528076171875,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/max_abs":0.000659942626953125,"train/train/tensor_act_model_layers_50_mlp_down_proj/norm":408.53799970398313,"train/train/tensor_act_model_layers_84_input_layernorm/mean":0.008270263671875,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/mean":1.3103708624839783e-06,"train/train/tensor_act_model_layers_42_mlp/max_abs":0.435546875,"train/train/tensor_act_model_layers_91_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/std":6.535928352476166e-05,"train/train/tensor_act_model_layers_62_self_attn_o_proj/max_abs":1.6484375,"train/train/tensor_act_model_layers_68_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/mean":-3.6612618714571e-08,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_0/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/mean":-2.95440258923918e-08,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/norm":0.02071152282831684,"train/train/tensor_act_model_layers_15/std":1.2793025169539347,"train/train/tensor_act_model_layers_6_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/std":2.7089238237648753e-05,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/std":0.051025390625,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/max_abs":0.11572265625,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/norm":0.0056405739623426295,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/mean":4.553794860839844e-05,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_mlp_gate_proj_weight/max_abs":0.000446319580078125,"train/train/tensor_act_model_layers_16_post_attention_layernorm/norm":5792.607299807768,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/mean":1.4796853065490723e-05,"train/train/tensor_param_model_layers_13_mlp_gate_proj_weight/max_abs":0.123046875,"train/train/tensor_param_model_layers_25_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/norm":0.01487393071634925,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/mean":9.66247171163559e-09,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/norm":4.75,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/mean":-1.6298145055770874e-09,"train/train/tensor_param_model_layers_85_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn_o_proj/std":0.02810918224100752,"train/train/tensor_grad_model_layers_44_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/mean":5.766749382019043e-06,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn_k_proj/mean":0.0068359375,"train/train/tensor_act_model_layers_93_mlp_up_proj/std":0.8398438043719098,"train/train/tensor_act_model_layers_69_mlp_down_proj/mean":-0.0005826950073242188,"train/train/tensor_act_model_layers_85_self_attn_q_proj/norm":6705.917835520313,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/mean":-0.00026702880859375,"train/train/tensor_act_model_layers_55_self_attn_v_proj/max_abs":2.046875,"train/train/tensor_grad_model_layers_68_mlp_gate_proj_weight/norm":0.025396933138792422,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_input_layernorm/norm":5792.614501957791,"train/train/tensor_param_model_layers_67_mlp_gate_proj_weight/std":0.037841796875,"train/train/layer__model_layers_20/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/norm":7.75,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/max_abs":0.00170135498046875,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/max_abs":0.0003643035888671875,"train/train/layer_model_layers_52/grad/std":4.9800408783715595e-05,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/mean":1.1594966053962708e-07,"train/train/tensor_act_model_layers_68_input_layernorm/mean":0.00658416748046875,"train/train/tensor_act_model_layers_81_self_attn_k_proj/norm":6235.975074044888,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/std":2.0970793686120872e-05,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/std":4.3756905692359806e-05,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/mean":-7.867813110351562e-05,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/mean":-0.00010204315185546875,"train/train/tensor_act_model_layers_41_mlp_up_proj/norm":2590.6269067326616,"train/train/tensor_grad_model_layers_48_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_o_proj/mean":0.001911163330078125,"train/train/tensor_act_model_layers_42_self_attn/norm":460.7705034637365,"train/train/tensor_act_model_layers_90_self_attn_k_proj/norm":6511.546461954347,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/std":4.6098261568166295e-05,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_q_proj/norm":5315.069643419208,"train/train/tensor_act_model_layers_83_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp/norm":1273.7648584228398,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_post_attention_layernorm/max_abs":5.96875,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp/mean":0.0008001327514648438,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/norm":0.02481550146153099,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46/mean":-0.0145263671875,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp_down_proj/max_abs":0.3984375,"train/train/tensor_param_model_layers_18_input_layernorm_weight/std":0,"train/train/layer__model_layers_32/param/std":0.05192495715011497,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/mean":-4.464387893676758e-05,"train/train/tensor_param_model_layers_22_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_mlp_gate_proj_weight/std":0.03955078125,"train/train/tensor_act_model_layers_76_post_attention_layernorm/mean":0.0049724578857421875,"train/train/tensor_param_model_layers_10_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/mean":1.4901161193847656e-05,"train/train/tensor_act_model_layers_72_self_attn_q_proj/norm":5786.089198205923,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/norm":5.78125,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/max_abs":0.000293731689453125,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp_up_proj/max_abs":2.390625,"train/train/tensor_act_model_layers_38_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/norm":5.53125,"train/train/tensor_act_model_layers_44_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_39/grad/max_abs":0.0022735595703125,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_mlp_up_proj/mean":0.0028247833251953125,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/std":0.03369140625,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_mlp_gate_proj_weight/mean":-0.0003509521484375,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/std":0.03173828125,"train/train/tensor_act_model_layers_43_input_layernorm/mean":-0.01385498046875,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/max_abs":0.000453948974609375,"train/train/tensor_act_model_layers_41_mlp_down_proj/mean":0.0005826950073242188,"train/train/tensor_act_model_layers_36_self_attn/max_abs":1.6953125,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/max_abs":0.2373046875,"train/train/tensor_act_model_layers_77_mlp/mean":0.0015316009521484375,"train/train/tensor_act_model_layers_42_self_attn_k_proj/mean":0.023529052734375,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/std":6.225857696918043e-05,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/max_abs":0.2001953125,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/mean":1.607550075277686e-07,"train/train/tensor_param_model_layers_36_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_49/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/mean":2.873130142688751e-07,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/mean":-0.00010156631469726562,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_78/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_mlp_gate_proj_weight/norm":0.020302149996803825,"train/train/tensor_act_model_layers_75_mlp_down_proj/norm":904.0612064637012,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_input_layernorm/std":1.0000001136213477,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/mean":-8.7738037109375e-05,"train/train/tensor_act_model_layers_56_self_attn_v_proj/mean":0.0001379847526550293,"train/train/tensor_param_model_layers_36_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/std":9.609172536309522e-05,"train/train/tensor_act_model_layers_0_self_attn/mean":-0.0009202957153320312,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/max_abs":0.000705718994140625,"train/train/layer__model_layers_63/param/norm":22.485759382329075,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_k_proj/max_abs":5.03125,"train/train/tensor_act_model_layers_15_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_k_proj/max_abs":5.40625,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/std":0.04833984375,"train/train/tensor_act_model_layers_87_post_attention_layernorm/max_abs":5.59375,"train/train/tensor_param_model_layers_62_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_9/grad/norm":0.04213254896778554,"train/train/tensor_act_model_layers_79_self_attn_o_proj/max_abs":1.828125,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/mean":-1.578591763973236e-07,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/std":0.05322265625,"train/train/tensor_act_model_layers_30_mlp_down_proj/max_abs":0.31640625,"train/train/tensor_act_model_layers_13_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/norm":0.02536119075439759,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/max_abs":0.1806640625,"train/train/layer_model_layers_68/grad/max_abs":0.0030517578125,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/max_abs":0.00017070770263671875,"train/train/tensor_act_model_layers_91_mlp_down_proj/std":0.5166044882847162,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/max_abs":0.00026702880859375,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/max_abs":0.000640869140625,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/norm":4.875,"train/train/tensor_param_model_layers_46_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_mlp_gate_proj_weight/norm":5.84375,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/norm":0.016669862162701526,"train/train/tensor_act_model_layers_42/std":1.2343754557113773,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_91/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/max_abs":0.1875,"train/train/tensor_act_model_layers_55_self_attn/mean":0.0003726482391357422,"train/train/tensor_act_model_layers_83_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_82_mlp_gate_proj_weight/mean":-0.00083160400390625,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/max_abs":0.000385284423828125,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/norm":9.5,"train/train/tensor_act_model_layers_72_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/mean":-1.2896634871140122e-08,"train/train/tensor_act_model_layers_66_self_attn_q_proj/max_abs":8,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn_o_proj/max_abs":1.265625,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/mean":1.2491364032030106e-07,"train/train/tensor_grad_model_layers_8_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/norm":0.003914682406315352,"train/train/tensor_act_model_layers_9_mlp/norm":227.8674351137486,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/max_abs":0.30078125,"train/train/tensor_act_model_layers_88_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/mean":7.756054401397705e-06,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/norm":4.875,"train/train/tensor_act_model_layers_28_self_attn_o_proj/std":0.0965610683317751,"train/train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_gate_proj_weight/std":0.031982421875,"train/train/tensor_act_model_layers_75_mlp_gate_proj/mean":0.0107269287109375,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/norm":0.0386371340904323,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/norm":6.625,"train/train/tensor_act_model_layers_42_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_mlp_gate_proj_weight/max_abs":0.00080108642578125,"train/train/tensor_param_model_layers_18_mlp_gate_proj_weight/norm":4.71875,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/max_abs":0.0013885498046875,"train/train/tensor_act_model_layers_61_mlp/mean":-6.628036499023438e-05,"train/train/tensor_act_model_layers_24_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/max_abs":0.197265625,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/max_abs":0.00021266937255859375,"train/train/tensor_grad_model_layers_6_mlp_gate_proj_weight/norm":0.01775640572892848,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/std":0.0291748046875,"train/train/tensor_grad_model_layers_80_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/max_abs":0.1806640625,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/std":0.0556640625,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_mlp_gate_proj_weight/mean":-6.849586497992277e-08,"train/train/tensor_grad_model_layers_6_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_input_layernorm/norm":5792.609375001553,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/mean":5.014589987695217e-08,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_15_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_37/grad/norm":0.042278857151180474,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/mean":1.2130476534366608e-07,"train/train/tensor_act_model_layers_19_self_attn_k_proj/std":0.8320358549357753,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_1/param/max_abs":1,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/norm":0.023197856266358195,"train/train/tensor_grad_model_layers_39_mlp_gate_proj_weight/norm":0.018261245476612946,"train/train/tensor_grad_model_layers_51_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33/max_abs":9.1875,"train/train/layer_model_layers_5/grad/mean":5.154420044883365e-09,"train/train/tensor_act_model_layers_26_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/std":1.6877557954066587e-05,"train/train/tensor_act_model_layers_68_mlp/max_abs":1.0859375,"train/train/tensor_act_model_layers_27_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/std":8.035954047368457e-05,"train/train/tensor_param_model_layers_23_mlp_gate_proj_weight/std":0.0264892578125,"train/train/tensor_act_model_layers_73_self_attn_k_proj/norm":4798.008404347793,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/std":0.0001296177247344596,"train/train/tensor_act_model_layers_29_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_post_attention_layernorm/max_abs":6.28125,"train/train/tensor_act_model_layers_32/max_abs":9.125,"train/train/layer__model_layers_44/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/std":2.509217932183852e-05,"train/train/tensor_act_model_layers_2_input_layernorm/std":1.0000001168809762,"train/train/tensor_grad_model_layers_40_mlp_gate_proj_weight/norm":0.017737933738679792,"train/train/tensor_act_model_layers_43_self_attn_k_proj/max_abs":5.40625,"train/train/tensor_act_model_layers_71_post_attention_layernorm/mean":0.006572723388671875,"train/train/tensor_act_model_layers_37_self_attn_k_proj/max_abs":4.875,"train/train/tensor_act_model_layers_30_mlp/max_abs":0.31640625,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_69/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn/norm":398.87546874903586,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/max_abs":0.1787109375,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_self_attn_v_proj/std":0.38476567965556313,"train/train/tensor_act_model_layers_3_mlp_down_proj/norm":682.7734066122232,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/max_abs":0.1728515625,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/norm":5.09375,"train/train/tensor_act_model_layers_20/std":1.2675842158637045,"train/train/tensor_act_model_layers_21_input_layernorm/mean":-0.023773193359375,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/mean":-4.698522388935089e-07,"train/train/layer_model_layers_60/grad/max_abs":0.000942230224609375,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/norm":6.8125,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/mean":9.823590517044067e-06,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/mean":1.0064104571938515e-07,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/norm":6.875,"train/train/layer_model_layers_55/grad/max_abs":0.000919342041015625,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/max_abs":0.1318359375,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/std":1.449979382300869e-05,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/norm":0.000597709624160445,"train/train/tensor_act_model_layers_26_input_layernorm/mean":-0.0157318115234375,"train/train/tensor_act_model_layers_7_mlp_down_proj/std":0.042542353154980526,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_48_mlp_down_proj/norm":379.68714743653817,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp/mean":-0.000885009765625,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/std":0.052001953125,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_gate_proj_weight/max_abs":0.1162109375,"train/train/tensor_act_model_layers_73_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_input_layernorm/norm":5792.610717775829,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/norm":0.007326729661967123,"train/train/layer_model_layers_52/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/max_abs":0.000881195068359375,"train/train/layer_model_layers_71/act/max_abs":11,"train/train/tensor_param_model_layers_3_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_11_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_mlp_gate_proj_weight/std":0.0260009765625,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/mean":-0.00016021728515625,"train/train/tensor_param_model_layers_38_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/std":0.02587890625,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/max_abs":0.000614166259765625,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/std":0.029541015625,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_mlp_gate_proj_weight/mean":-0.00010728836059570312,"train/train/tensor_act_model_layers_72_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_gate_proj/norm":6366.415623397616,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/max_abs":0.142578125,"train/train/tensor_act_model_layers_2_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_mlp_gate_proj_weight/max_abs":0.11279296875,"train/train/tensor_act_model_layers_63_mlp_up_proj/max_abs":2.328125,"train/train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp_down_proj/mean":5.103647708892822e-06,"train/train/tensor_act_model_layers_46_mlp/std":0.06384313031042402,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/max_abs":0.0003643035888671875,"train/train/layer_model_layers_36/grad/norm":0.041900671318493905,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_26/act/norm":13773.673600985036,"train/train/layer_model_layers_23/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/norm":5.78125,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/norm":0.013185218428356483,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/norm":6.1875,"train/train/tensor_act_model_layers_69_mlp/std":0.13476564047817538,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_6/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/max_abs":0.000354766845703125,"train/train/tensor_act_model_layers_75_post_attention_layernorm/mean":0.00670623779296875,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_13/act/mean":-0.007612662655966622,"train/train/tensor_act_model_layers_64_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_input_layernorm/norm":5792.616577148977,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/std":0.0419921875,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/max_abs":0.20703125,"train/train/tensor_act_model_layers_91_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/std":0.0299072265625,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/norm":6.0625,"train/train/tensor_act_model_layers_19_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_51/param/std":0.05305341241913118,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/norm":4.625,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/mean":1.3940734788775444e-07,"train/train/tensor_act_model_layers_56_self_attn_v_proj/max_abs":2.40625,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_post_attention_layernorm/std":1.000000549945829,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/max_abs":0.000885009765625,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/std":0.03173828125,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/norm":0.013831299024202848,"train/train/tensor_act_model_layers_48_self_attn/mean":-0.0008687973022460938,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/mean":-6.67572021484375e-05,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/std":1.163338713571423e-05,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/max_abs":0.000492095947265625,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/max_abs":0.0003223419189453125,"train/train/tensor_grad_model_layers_60_mlp_gate_proj_weight/mean":-1.1990778148174286e-08,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/max_abs":0.1640625,"train/train/tensor_act_model_layers_89_self_attn_o_proj/std":0.17675783966957934,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/norm":5.65625,"train/train/tensor_act_model_layers_83_self_attn_k_proj/max_abs":6.375,"train/train/tensor_param_model_layers_22_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_input_layernorm/norm":5792.603881836549,"train/train/tensor_grad_model_layers_35_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/norm":0.0038216489392826148,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_37_mlp_down_proj/max_abs":0.494140625,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/std":0.0269775390625,"train/train/tensor_grad_model_layers_85_mlp_gate_proj_weight/max_abs":0.000492095947265625,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp/norm":320.6065601583395,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/mean":-1.5902332961559296e-07,"train/train/tensor_act_model_layers_56_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_q_proj/std":0.9453125236448174,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_post_attention_layernorm/norm":5792.608886720098,"train/train/tensor_act_model_layers_34_mlp_up_proj/norm":2331.1558171085935,"train/train/tensor_act_model_layers_32_self_attn_q_proj/std":0.9179688210182975,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/max_abs":0.00018024444580078125,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_17/act/frac_near_user_limit":0,"train/train/layer__model_layers_27/param/max_abs":1,"train/train/tensor_param_model_layers_44_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/mean":0.00066375732421875,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_23_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_gate_proj/norm":4048.50043639459,"train/train/tensor_act_model_layers_71_self_attn/max_abs":2.03125,"train/train/tensor_grad_model_layers_0_mlp_gate_proj_weight/max_abs":0.00494384765625,"train/train/tensor_grad_model_layers_62_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/max_abs":0.2158203125,"train/train/tensor_act_model_layers_82_input_layernorm/mean":0.00595855712890625,"train/train/tensor_act_model_layers_14_mlp_up_proj/max_abs":1.7890625,"train/train/tensor_act_model_layers_5/norm":7505.027045919865,"train/train/tensor_act_model_layers_18_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/std":6.677141411977031e-05,"train/train/tensor_act_model_layers_61_mlp/std":0.10351565774509885,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/mean":-3.7965946830809116e-08,"train/train/tensor_act_model_layers_36_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71/std":1.4804761184274111,"train/train/tensor_grad_model_layers_19_mlp_gate_proj_weight/max_abs":0.0005645751953125,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/mean":-0.0002593994140625,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/norm":0.00784917508072626,"train/train/layer__model_layers_38/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_post_attention_layernorm/max_abs":6.71875,"train/train/tensor_act_model_layers_89_post_attention_layernorm/std":1.000000861560322,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/max_abs":0.0010528564453125,"train/train/tensor_act_model_layers_73_mlp/std":0.14306742935027936,"train/train/tensor_act_model_layers_18_self_attn_v_proj/max_abs":1.890625,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/norm":0.019147439638842818,"train/train/tensor_act_model_layers_85_self_attn_v_proj/max_abs":2.765625,"train/train/tensor_param_model_layers_35_mlp_gate_proj_weight/std":0.029052734375,"train/train/layer_model_layers_16/grad/max_abs":0.00124359130859375,"train/train/tensor_grad_model_layers_48_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_o_proj/mean":-0.00011797621846199036,"train/train/layer_model_layers_44/act/norm":13651.114842562141,"train/train/layer_model_layers_38/act/mean":-0.007182772670473371,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_input_layernorm/norm":5792.61499023757,"train/train/tensor_act_model_layers_68_mlp_gate_proj/norm":3561.486771835925,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/std":0.04296875,"train/train/tensor_act_model_layers_10_input_layernorm/std":1.0000004023312714,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/max_abs":0.000255584716796875,"train/train/tensor_act_model_layers_36_self_attn_k_proj/norm":5091.505219337561,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/norm":4.9375,"train/train/tensor_act_model_layers_61_mlp_gate_proj/norm":3328.213293181494,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_mlp_gate_proj_weight/norm":4.75,"train/train/tensor_act_model_layers_93_post_attention_layernorm/norm":5792.613281254218,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/mean":1.232256181538105e-07,"train/train/tensor_act_model_layers_51_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/max_abs":0.212890625,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_65/act/std":0.6989809980116712,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/norm":5.75,"train/train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/norm":0.0006568040247630455,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/mean":4.842877388000488e-06,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/std":2.7452908715489203e-05,"train/train/tensor_act_model_layers_60_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/mean":-1.4975666999816895e-06,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_mlp_gate_proj_weight/mean":-6.583286449313164e-08,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/max_abs":0.1875,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/std":2.7495460960205184e-05,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/std":5.87785605818655e-05,"train/train/tensor_act_model_layers_13_mlp_up_proj/std":0.23730483577570582,"train/train/tensor_act_model_layers_91_mlp_up_proj/max_abs":4.40625,"train/train/tensor_act_model_layers_29_mlp_gate_proj/std":0.27343772780140685,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/norm":4.5625,"train/train/tensor_act_model_layers_7/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/std":0.03271484375,"train/train/tensor_act_model_layers_50_mlp_gate_proj/max_abs":2.21875,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/std":0.039794921875,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/max_abs":0.000659942626953125,"train/train/tensor_act_model_layers_90_mlp_gate_proj/max_abs":4.15625,"train/train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/norm":0.006593033003581549,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/mean":-3.6670826375484467e-08,"train/train/tensor_act_model_layers_71_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_o_proj/mean":-0.0006208419799804688,"train/train/tensor_act_model_layers_40/norm":7180.659063512985,"train/train/tensor_act_model_layers_36/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_o_proj/mean":0.003040313720703125,"train/train/layer_model_layers_68/grad/norm":0.07402763515997243,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/max_abs":9.965896606445312e-05,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/std":2.9682298413477503e-05,"train/train/tensor_act_model_layers_42_mlp_down_proj/max_abs":0.435546875,"train/train/tensor_act_model_layers_1_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_mlp_gate_proj_weight/mean":0.00048828125,"train/train/tensor_act_model_layers_54_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/std":2.9501313815071945e-05,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_6/act/max_abs":8.4375,"train/train/tensor_act_model_layers_17/max_abs":8.375,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/max_abs":0.10400390625,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp/norm":433.7878417626019,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/norm":4.84375,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/std":3.740877903845874e-05,"train/train/tensor_act_model_layers_70_mlp_up_proj/mean":0.0019168853759765625,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/max_abs":0.248046875,"train/train/tensor_act_model_layers_90_input_layernorm/norm":5792.607788086466,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/norm":5.125,"train/train/tensor_param_model_layers_63_mlp_gate_proj_weight/norm":6.65625,"train/train/tensor_act_model_layers_39_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_28_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/max_abs":0.000301361083984375,"train/train/tensor_act_model_layers_89_post_attention_layernorm/norm":5792.608032228243,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/norm":0.02878511409520896,"train/train/tensor_act_model_layers_20_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/norm":6.0625,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/mean":-0.0002727508544921875,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/mean":1.2913369573652744e-07,"train/train/tensor_act_model_layers_21_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_8/grad/norm":0.03787484451614237,"train/train/layer_model_layers_79/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_mlp_gate_proj_weight/norm":4.71875,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/mean":-2.6226043701171875e-05,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/norm":6.78125,"train/train/tensor_act_model_layers_16_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_55/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/norm":5.75,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/std":0.05810546875,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/mean":0.00014019012451171875,"train/train/tensor_act_model_layers_37_mlp_gate_proj/max_abs":1.8984375,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/max_abs":0.18359375,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/max_abs":0.00031280517578125,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/std":4.5536350714240654e-05,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/norm":0.019026822801356782,"train/train/tensor_act_model_layers_21_self_attn_v_proj/max_abs":2.53125,"train/train/layer_model_layers_76/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp_gate_proj/mean":0.00049591064453125,"train/train/tensor_act_model_layers_13_self_attn_v_proj/std":0.285645829704619,"train/train/tensor_act_model_layers_13_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/max_abs":0.15625,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/max_abs":0.000377655029296875,"train/train/tensor_act_model_layers_54_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_k_proj/norm":4549.361307185045,"train/train/tensor_act_model_layers_47_mlp_gate_proj/norm":2736.5656688962067,"train/train/tensor_act_model_layers_10_post_attention_layernorm/norm":5792.605346681852,"train/train/layer__model_layers_9/param/mean":0.0016243766510914343,"train/train/tensor_param_model_layers_67_mlp_gate_proj_weight/max_abs":0.1669921875,"train/train/tensor_param_model_layers_58_mlp_gate_proj_weight/std":0.03515625,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/std":1.4896566468227187e-05,"train/train/layer__model_layers_60/param/mean":0.0014320885335413417,"train/train/tensor_act_model_layers_20_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_mlp_down_proj/std":0.03936784714423077,"train/train/tensor_act_model_layers_54_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_75_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/max_abs":0.00021457672119140625,"train/train/tensor_act_model_layers_11_self_attn_v_proj/std":0.32031254295441147,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_gate_proj_weight/std":6.463766001670986e-05,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp_down_proj/mean":-0.00052642822265625,"train/train/tensor_act_model_layers_54_post_attention_layernorm/mean":-0.002460479736328125,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/norm":0.02073246658040068,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_post_attention_layernorm/mean":0.00659942626953125,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/norm":3.375,"train/train/layer_model_layers_31/act/mean":-9.764358401298523e-05,"train/train/tensor_param_model_layers_1_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/mean":1.4805118553340435e-07,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/std":0.02685546875,"train/train/tensor_act_model_layers_66_mlp_up_proj/norm":3504.6515320848384,"train/train/tensor_act_model_layers_29_self_attn_v_proj/max_abs":2.34375,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/norm":5.09375,"train/train/tensor_param_model_layers_18_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_73_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/max_abs":0.1083984375,"train/train/layer_model_layers_54/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19/std":1.2714904262762032,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_post_attention_layernorm/norm":5792.602172863039,"train/train/tensor_act_model_layers_90_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_post_attention_layernorm/norm":5792.600219732349,"train/train/tensor_act_model_layers_40_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72/norm":8671.125007625174,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_mlp_gate_proj_weight/max_abs":0.248046875,"train/train/tensor_act_model_layers_87/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_input_layernorm/mean":0.005542755126953125,"train/train/layer_model_layers_86/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/max_abs":0.08154296875,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/mean":0.00032806396484375,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/max_abs":0.00165557861328125,"train/train/tensor_act_model_layers_31_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp/mean":0.00019049644470214844,"train/train/tensor_act_model_layers_86/norm":11685.477011483259,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/max_abs":0.1640625,"train/train/tensor_act_model_layers_30_mlp_down_proj/mean":0.0005731582641601562,"train/train/tensor_grad_model_layers_67_mlp_gate_proj_weight/std":7.570348296305383e-05,"train/train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/mean":-3.837631084024906e-07,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/mean":4.906207323074341e-06,"train/train/tensor_act_model_layers_12_input_layernorm/std":1.0000004037282546,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/std":0.04638671875,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/max_abs":0.169921875,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/std":5.968514578829783e-05,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/max_abs":0.00029754638671875,"train/train/layer_model_layers_84/grad/mean":-2.1247778108921746e-07,"train/train/tensor_act_model_layers_86_post_attention_layernorm/norm":5792.611816411734,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/norm":2.828125,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/mean":-0.000255584716796875,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/mean":1.914449967443943e-07,"train/train/layer_model_layers_12/act/max_abs":8.5,"train/train/tensor_act_model_layers_15_mlp_up_proj/std":0.24096737002631308,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/std":1.6137467044252913e-05,"train/train/tensor_act_model_layers_55_self_attn_o_proj/mean":0.0003726482391357422,"train/train/tensor_grad_model_layers_93_mlp_gate_proj_weight/max_abs":0.000881195068359375,"train/train/layer_model_layers_6/grad/mean":1.3130676660936633e-07,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/norm":0.013054569568021451,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/norm":4.96875,"train/train/layer_model_layers_3/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/norm":6.1875,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/max_abs":9.72747802734375e-05,"train/train/tensor_grad_model_layers_41_mlp_gate_proj_weight/std":4.919216570958604e-05,"train/train/tensor_act_model_layers_24_self_attn_q_proj/std":0.8828168272272646,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/norm":0.01477914352811313,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn_v_proj/std":0.3891611290683816,"train/train/layer__model_layers_21/param/norm":20.48565618609763,"train/train/layer_model_layers_20/act/norm":13255.468460979328,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/norm":0.0038988594833021047,"train/train/tensor_act_model_layers_37_self_attn_o_proj/mean":-0.0011739730834960938,"train/train/layer_model_layers_91/act/mean":0.012078148978097098,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/std":0.0341796875,"train/train/tensor_act_model_layers_89_mlp/max_abs":2.421875,"train/train/tensor_act_model_layers_17_input_layernorm/mean":-0.02789306640625,"train/train/tensor_act_model_layers_1_mlp/max_abs":1.7421875,"train/train/tensor_act_model_layers_92_post_attention_layernorm/mean":0.012176513671875,"train/train/tensor_grad_model_norm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_6/param/max_abs":1,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/max_abs":0.12109375,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/std":0.033935546875,"train/train/tensor_param_model_layers_23_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/norm":4.03125,"train/train/layer_model_layers_14/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/max_abs":0.0004596710205078125,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp_gate_proj/std":0.6992191577422696,"train/train/tensor_act_model_layers_2_self_attn/std":0.038760165551908504,"train/train/tensor_act_model_layers_43/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_k_proj/max_abs":4.6875,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/max_abs":0.0003376007080078125,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/mean":0.0006561279296875,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/mean":1.2069940567016602e-06,"train/train/tensor_act_model_layers_63_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/std":7.048112522851776e-05,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_43/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58/std":1.3046883731547045,"train/train/layer__model_layers_45/param/mean":0.0015062162545095555,"train/train/layer_model_layers_55/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/std":0.064453125,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/norm":0.004547759145461596,"train/train/tensor_act_model_layers_39_mlp/norm":327.20775240231006,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/max_abs":0.00133514404296875,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/norm":6.03125,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/norm":0.034070521818732574,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp_down_proj/max_abs":1.078125,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/max_abs":0.13671875,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp/max_abs":1.1953125,"train/train/tensor_param_model_layers_36_mlp_gate_proj_weight/max_abs":0.130859375,"train/train/tensor_act_model_layers_70_self_attn_v_proj/std":0.41259863991944584,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/mean":-8.726119995117188e-05,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/max_abs":0.0013580322265625,"train/train/tensor_act_model_layers_63/norm":7751.419580655314,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/max_abs":0.150390625,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/norm":3.234375,"train/train/tensor_act_model_layers_73_self_attn/std":0.12281141495276253,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/max_abs":0.1318359375,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/norm":0.016452456302222508,"train/train/tensor_param_model_layers_53_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_47_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/std":8.371568906918423e-05,"train/train/tensor_param_model_layers_50_mlp_gate_proj_weight/max_abs":0.173828125,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/norm":0.0032924086144119663,"train/train/tensor_act_model_layers_7_input_layernorm/std":1.0000001937150769,"train/train/tensor_act_model_layers_54_self_attn_k_proj/norm":5157.163765970099,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/mean":-1.4126300811767578e-05,"train/train/tensor_param_model_layers_41_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_31_self_attn_o_proj/max_abs":0.8984375,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/norm":0.0018296037329316337,"train/train/tensor_act_model_layers_61/norm":7670.079270756721,"train/train/tensor_act_model_layers_88_self_attn_k_proj/std":1.0078281726653513,"train/train/tensor_param_model_layers_59_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/std":6.492889642651609e-05,"train/train/tensor_act_model_layers_28_self_attn_q_proj/std":0.991212418277931,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/max_abs":0.000232696533203125,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/std":4.257036951890683e-05,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/std":0.02734375,"train/train/tensor_grad_model_layers_66_mlp_gate_proj_weight/norm":0.02792723584919003,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/mean":-5.517154932022095e-06,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_q_proj/max_abs":7.84375,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/norm":0.013726974900140115,"train/train/tensor_act_model_layers_47_self_attn_o_proj/mean":6.048381328582764e-05,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/norm":8.625,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/mean":-1.328880898654461e-07,"train/train/layer_model_layers_56/grad/max_abs":0.00188446044921875,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/std":6.176591147111695e-05,"train/train/tensor_act_model_layers_3_self_attn_o_proj/std":0.02780239516545149,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_up_proj/mean":0.0027313232421875,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/std":3.403236284692661e-05,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/mean":-1.125037670135498e-06,"train/train/tensor_act_model_layers_63_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/max_abs":0.1552734375,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/mean":0.0002956390380859375,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_73/param/mean":0.0015254028129875195,"train/train/tensor_act_model_layers_11_mlp/max_abs":0.7265625,"train/train/tensor_param_model_layers_66_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19/mean":-0.030609130859375,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/max_abs":0.2578125,"train/train/tensor_param_model_layers_73_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/norm":6.6875,"train/train/tensor_grad_model_layers_70_mlp_gate_proj_weight/std":7.08226624055138e-05,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_input_layernorm/norm":5792.613891604684,"train/train/layer_model_layers_84/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/norm":0.0006343819760980757,"train/train/tensor_act_model_layers_47_self_attn/norm":554.7975450313287,"train/train/tensor_act_model_layers_84_mlp/mean":-0.00225830078125,"train/train/tensor_act_model_layers_83_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn_o_proj/mean":0.0006527900695800781,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/max_abs":0.000308990478515625,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/max_abs":0.000705718994140625,"train/train/tensor_act_model_layers_1/norm":7404.03190624453,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_53_self_attn_o_proj/max_abs":1.7578125,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/std":0.0380859375,"train/train/tensor_act_model_layers_81_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_mlp_gate_proj_weight/max_abs":0.00061798095703125,"train/train/tensor_act_model_layers_77_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_post_attention_layernorm/norm":5792.6119384794165,"train/train/tensor_act_model_layers_31_mlp/norm":256.3926885119202,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/mean":1.792795956134796e-08,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/max_abs":0.0014190673828125,"train/train/tensor_act_model_layers_72_self_attn_k_proj/max_abs":6.65625,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/std":7.403781797314318e-05,"train/train/tensor_act_model_layers_59_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_rotary_emb/mean":0.333984375,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_14_self_attn_o_proj/mean":0.0008778572082519531,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/std":0.00010811095574226847,"train/train/layer_model_layers_7/act/norm":13671.835256784356,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/mean":0.00130462646484375,"train/train/tensor_act_model_layers_35_mlp_down_proj/max_abs":0.435546875,"train/train/tensor_act_model_layers_4_self_attn/norm":798.2307543961442,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/norm":7.34375,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/norm":0.015541149695425657,"train/train/tensor_act_model_layers_76_mlp/max_abs":1.1171875,"train/train/tensor_act_model_layers_90_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_v_proj/mean":-0.00296783447265625,"train/train/tensor_param_model_layers_34_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_norm/max_abs":5.09375,"train/train/layer__model_layers_89/param/norm":25.221537949984334,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/std":0.051513671875,"train/train/tensor_act_model_layers_28/max_abs":9,"train/train/tensor_grad_model_layers_1_mlp_gate_proj_weight/std":9.574099752335628e-05,"train/train/tensor_param_model_layers_51_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/std":4.591575926568555e-05,"train/train/tensor_act_model_layers_50/mean":-0.009979248046875,"train/train/tensor_act_model_layers_64_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/max_abs":0.0002422332763671875,"train/train/tensor_act_model_layers_2_self_attn/max_abs":0.76171875,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/norm":0.0010839648717551934,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/mean":-0.000629425048828125,"train/train/tensor_act_model_layers_63/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_self_attn/std":0.10095430427195932,"train/train/layer_model_layers_25/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn/norm":406.2135493347889,"train/train/tensor_act_model_layers_93_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/std":0.055908203125,"train/train/layer_model_layers_66/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/max_abs":1.28125,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/mean":-0.00010776519775390625,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/max_abs":0.1630859375,"train/train/tensor_act_model_layers_72_self_attn_o_proj/std":0.154543290292291,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_down_proj/norm":273.00844885139077,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/norm":0.0010256021287179781,"train/train/tensor_act_model_layers_89_self_attn/mean":8.660554885864258e-05,"train/train/tensor_act_model_layers_73_mlp_down_proj/std":0.14306742935027936,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_66/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_mlp_gate_proj_weight/std":4.192010892812712e-05,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/mean":3.1781382858753204e-07,"train/train/tensor_act_model_layers_87_self_attn_k_proj/mean":-0.032470703125,"train/train/layer__model_layers_73/param/std":0.05780163070322878,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/norm":0.023309277426805292,"train/train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/mean":4.8354268074035645e-06,"train/train/tensor_act_model_layers_32_mlp_down_proj/mean":0.00014793872833251953,"train/train/tensor_act_model_layers_39_self_attn/mean":-0.00208282470703125,"train/train/layer__model_layers_50/param/max_abs":1,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/norm":0.0006775679844412262,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/max_abs":0.0017547607421875,"train/train/tensor_act_model_layers_36_self_attn_v_proj/max_abs":3.03125,"train/train/tensor_act_model_layers_11_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn_o_proj/std":0.0738554636007003,"train/train/layer_model_layers_43/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_o_proj/norm":534.4974967848466,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/max_abs":0.11279296875,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/max_abs":0.212890625,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_87/grad/std":8.959635127223893e-05,"train/train/tensor_act_model_layers_78_self_attn_q_proj/max_abs":6.40625,"train/train/tensor_grad_model_layers_73_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/max_abs":0.25,"train/train/layer__model_layers_47/param/std":0.053504416196531034,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/max_abs":0.0030517578125,"train/train/tensor_act_model_layers_67_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/max_abs":0.171875,"train/train/tensor_act_model_layers_76_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_o_proj/mean":0.0003173351287841797,"train/train/layer_model_layers_77/grad/mean":-3.5012080573858967e-07,"train/train/layer__model_layers_66/param/std":0.05827633707703121,"train/train/tensor_act_model_layers_87_mlp/std":0.3125001600710265,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/mean":8.925795555114746e-06,"train/train/tensor_act_model_layers_42_self_attn/max_abs":1.1484375,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/mean":-3.1210947781801224e-07,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/mean":2.3748725652694702e-07,"train/train/tensor_grad_model_layers_15_mlp_gate_proj_weight/mean":1.8277205526828766e-07,"train/train/tensor_act_model_layers_88_mlp/mean":0.0058441162109375,"train/train/tensor_act_model_layers_76_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn_o_proj/norm":400.6547887434907,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/std":0.03271484375,"train/train/tensor_act_model_layers_37_self_attn_q_proj/std":1.08985080297984,"train/train/tensor_act_model_layers_75_post_attention_layernorm/norm":5792.611816407506,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/mean":-6.3478946685791016e-06,"train/train/tensor_act_model_layers_42_post_attention_layernorm/std":1.000000309664708,"train/train/tensor_act_model_layers_19_self_attn_v_proj/mean":-0.00269317626953125,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/std":0.04736328125,"train/train/tensor_act_model_layers_78_mlp_up_proj/mean":-0.0105133056640625,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/max_abs":0.0003643035888671875,"train/train/tensor_act_model_layers_60_self_attn/std":0.08874606221098551,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/mean":-4.336470738053322e-08,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/norm":6.25,"train/train/tensor_act_model_layers_2_self_attn/mean":-0.0003018379211425781,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_o_proj/std":0.19165096657774267,"train/train/tensor_act_model_layers_63/mean":0.0014019012451171875,"train/train/tensor_param_model_layers_71_mlp_gate_proj_weight/max_abs":0.1787109375,"train/train/tensor_param_model_layers_91_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25/max_abs":8.75,"train/train/tensor_act_model_layers_57_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/std":0.042724609375,"train/train/layer_model_layers_24/act/mean":-0.0025260043995721,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/max_abs":0.0013275146484375,"train/train/tensor_act_model_layers_50_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/max_abs":0.00058746337890625,"train/train/tensor_act_model_layers_78_self_attn_v_proj/std":0.418946549259819,"train/train/layer_model_layers_46/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp/max_abs":0.7421875,"train/train/layer_model_layers_27/act/mean":-0.004380771092006138,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/norm":5.75,"train/train/tensor_act_model_layers_64_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/std":5.503463816952431e-05,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/norm":0.0016601166299692314,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_29/act/mean":-0.010203361511230469,"train/train/tensor_param_model_layers_54_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_gate_proj_weight/max_abs":0.00070953369140625,"train/train/tensor_param_model_layers_72_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_37/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_q_proj/mean":0.010589599609375,"train/train/tensor_act_model_layers_32_post_attention_layernorm/mean":-0.012603759765625,"train/train/tensor_act_model_layers_86_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/norm":6.65625,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/mean":2.48473952524364e-08,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/std":4.20205524551943e-05,"train/train/tensor_param_model_layers_58_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_61/act/max_abs":9.6875,"train/train/tensor_act_model_layers_65_self_attn_v_proj/norm":2663.5192569230717,"train/train/tensor_act_model_layers_81_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_down_proj/std":0.0564576209557112,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/norm":5.3125,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/max_abs":0.00013065338134765625,"train/train/layer__model_layers_60/param/std":0.05550042050765796,"train/train/tensor_act_model_layers_32_post_attention_layernorm/max_abs":6.71875,"train/train/tensor_act_model_layers_31_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_o_proj/norm":428.1534843436557,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_o_proj/norm":956.0093340802019,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/max_abs":1,"train/train/layer__model_layers_29/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/mean":-3.434251993894577e-07,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_4_self_attn_k_proj/max_abs":7.6875,"train/train/tensor_act_model_layers_58_self_attn_k_proj/norm":4879.520304785125,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/norm":0.005302028503646592,"train/train/tensor_param_model_layers_23_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_81/param/mean":0.0017129835584048362,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/norm":0.027192720482007694,"train/train/tensor_act_model_layers_66_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/mean":3.577442839741707e-07,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/max_abs":0.000499725341796875,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/norm":0.0015537857149161307,"train/train/tensor_act_model_layers_14_self_attn_k_proj/max_abs":4.875,"train/train/tensor_act_model_layers_50_mlp_gate_proj/norm":2830.165829631439,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/mean":-0.000614166259765625,"train/train/tensor_act_model_layers_74_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90/frac_near_dtype_limit":0,"train/train/layer__model_layers_43/param/max_abs":1,"train/train/layer__model_layers_86/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_post_attention_layernorm/mean":0.0125732421875,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/norm":0.00240325927734375,"train/train/tensor_act_model_layers_77_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/mean":1.4237593859434128e-07,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/mean":-2.872943878173828e-05,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/mean":3.022141754627228e-07,"train/train/tensor_param_model_layers_82_mlp_gate_proj_weight/std":0.04296875,"train/train/layer__model_layers_24/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/mean":0.000476837158203125,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/mean":5.266629159450531e-07,"train/train/tensor_act_model_layers_53_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_49/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/std":8.592216476080322e-05,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_input_layernorm/std":1.000000764964787,"train/train/tensor_act_model_layers_93_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/mean":-0.00015926361083984375,"train/train/layer_model_layers_33/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_86/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_q_proj/std":1.0195389839587883,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/max_abs":0.185546875,"train/train/layer__model_layers_0/param/norm":21.054425149107608,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/std":8.916348174231589e-05,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/norm":0.0008037115409334311,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/mean":3.0517578125e-05,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_gate_proj_weight/norm":5.8125,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/std":0.04150390625,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/std":0.025146484375,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/max_abs":0.1533203125,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/max_abs":0.0003795623779296875,"train/train/tensor_act_model_layers_80_mlp_down_proj/mean":0.0008382797241210938,"train/train/tensor_act_model_layers_87_self_attn/mean":0.001373291015625,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/std":1.7598534790545405e-05,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/std":0.037841796875,"train/train/tensor_act_model_layers_24_mlp_gate_proj/std":0.2480470818561908,"train/train/tensor_grad_model_layers_82_mlp_gate_proj_weight/max_abs":0.00057220458984375,"train/train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer__model_layers_24/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_23_mlp_gate_proj_weight/max_abs":0.1259765625,"train/train/layer_model_layers_72/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_v_proj/std":0.2846692538955737,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/mean":-2.8725480660796165e-07,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn/mean":-0.0011739730834960938,"train/train/tensor_act_model_layers_24_mlp/std":0.036865237286153894,"train/train/tensor_act_model_layers_71_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/mean":-0.000682830810546875,"train/train/tensor_act_model_layers_89_post_attention_layernorm/mean":0.00829315185546875,"train/train/tensor_grad_model_layers_42_mlp_gate_proj_weight/std":4.990898902636432e-05,"train/train/tensor_act_model_layers_12_self_attn_k_proj/max_abs":5.15625,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_44/grad/std":4.674262983119928e-05,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/mean":-7.486343383789062e-05,"train/train/layer_model_layers_6/grad/std":6.054959241519475e-05,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/std":0.05419921875,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_gate_proj_weight/max_abs":0.001800537109375,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/norm":7,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/std":0.042236328125,"train/train/layer_model_layers_23/act/max_abs":8.5,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/max_abs":0.15234375,"train/train/tensor_act_model_layers_16_mlp_down_proj/mean":0.0006818771362304688,"train/train/tensor_param_model_layers_85_mlp_gate_proj_weight/norm":8.1875,"train/train/tensor_act_model_layers_11_mlp_gate_proj/max_abs":1.6640625,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/mean":-1.123407855629921e-08,"train/train/tensor_act_model_layers_18_self_attn/max_abs":0.56640625,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/mean":2.894375938922167e-07,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_up_proj/mean":0.012115478515625,"train/train/tensor_act_model_layers_51_input_layernorm/max_abs":6.0625,"train/train/tensor_act_model_layers_26_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_13/param/mean":0.0015552836163739518,"train/train/tensor_act_model_layers_8_mlp_up_proj/max_abs":2.078125,"train/train/tensor_grad_model_layers_71_mlp_gate_proj_weight/norm":0.02618827875112907,"train/train/layer_model_layers_1/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn_q_proj/max_abs":5.78125,"train/train/tensor_act_model_layers_55_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_post_attention_layernorm/mean":-0.029998779296875,"train/train/layer_model_layers_91/grad/std":9.065031177630174e-05,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/std":6.760645395134425e-05,"train/train/tensor_grad_model_layers_29_mlp_gate_proj_weight/mean":-1.8568243831396103e-07,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/mean":-3.762543201446533e-07,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/norm":3.15625,"train/train/layer__model_layers_55/param/max_abs":1,"train/train/tensor_act_model_layers_11/mean":-0.03192138671875,"train/train/layer_model_layers_42/act/std":0.6380025111978548,"train/train/tensor_act_model_layers_23_self_attn_q_proj/max_abs":5.96875,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/std":0.05224609375,"train/train/tensor_act_model_layers_21_mlp_up_proj/norm":2026.596388595696,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_mlp_gate_proj_weight/mean":3.743171691894531e-05,"train/train/tensor_act_model_layers_90_mlp/std":0.4199219106934776,"train/train/tensor_act_model_layers_41_mlp/max_abs":0.470703125,"train/train/tensor_act_model_layers_83_self_attn_v_proj/std":0.4062500094565061,"train/train/tensor_param_model_layers_19_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_input_layernorm/mean":-0.0174560546875,"train/train/tensor_param_model_layers_77_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/std":0.04443359375,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/mean":0.00018405914306640625,"train/train/tensor_param_model_layers_58_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/std":0.054931640625,"train/train/tensor_act_model_layers_42_self_attn_k_proj/max_abs":5.71875,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/mean":-7.677078247070312e-05,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_post_attention_layernorm/norm":5792.609252931718,"train/train/layer__model_layers_88/param/mean":0.0015857000246806748,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_q_proj/norm":7653.719048745098,"train/train/tensor_grad_model_layers_31_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_o_proj/mean":-0.00015032291412353516,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/std":2.8944149476133734e-05,"train/train/tensor_act_model_layers_54_post_attention_layernorm/norm":5792.606567387869,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/std":0.0001672958722723539,"train/train/tensor_param_model_layers_29_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_input_layernorm/std":1.0000008208440971,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/max_abs":0.000640869140625,"train/train/tensor_act_model_layers_24_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_post_attention_layernorm/mean":-0.028350830078125,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/std":0.0284423828125,"train/train/layer_model_layers_68/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/mean":1.62515789270401e-07,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/mean":-0.00021457672119140625,"train/train/tensor_act_model_layers_29_self_attn_o_proj/std":0.08105543700923366,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/norm":0.0059759737796915985,"train/train/tensor_param_model_layers_76_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/mean":-8.761882781982422e-06,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_87/act/std":0.8811760826043425,"train/train/tensor_act_model_layers_61_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/std":4.851734605345868e-05,"train/train/tensor_act_model_layers_18/norm":7371.272505236948,"train/train/tensor_act_model_layers_43_post_attention_layernorm/mean":-0.0128631591796875,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_48/grad/mean":4.2474702256332134e-08,"train/train/tensor_act_model_layers_40_input_layernorm/norm":5792.607177737271,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_mlp_gate_proj_weight/max_abs":0.00048065185546875,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/norm":0.003456343081240113,"train/train/tensor_param_model_layers_89_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/mean":-3.3527612686157227e-08,"train/train/tensor_act_model_layers_4_post_attention_layernorm/max_abs":4.90625,"train/train/tensor_act_model_layers_8_self_attn_k_proj/mean":-0.0196533203125,"train/train/tensor_act_model_layers_22_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_mlp_gate_proj_weight/mean":3.719329833984375e-05,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn/std":0.05798357486060793,"train/train/tensor_act_model_layers_30_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/mean":4.556030035018921e-06,"train/train/layer_model_layers_93/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7/max_abs":8.75,"train/train/tensor_act_model_layers_73_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_post_attention_layernorm/mean":0.00783538818359375,"train/train/tensor_param_model_layers_10_mlp_gate_proj_weight/norm":4.5,"train/train/tensor_act_model_layers_22_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/mean":-9.16188582777977e-08,"train/train/tensor_act_model_layers_87_self_attn_v_proj/max_abs":2.71875,"train/train/tensor_act_model_layers_85/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/norm":0.0006130219692759467,"train/train/tensor_act_model_layers_42_mlp_up_proj/std":0.32031257759506165,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/std":0.055908203125,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_49_mlp/norm":395.6589739827596,"train/train/tensor_act_model_layers_80_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_34/act/mean":0.0020372612135750906,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_52/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/max_abs":0.0012359619140625,"train/train/tensor_act_model_layers_47_self_attn/max_abs":1.40625,"train/train/tensor_act_model_layers_52_self_attn/norm":241.709215287855,"train/train/tensor_act_model_layers_76/frac_near_dtype_limit":0,"train/train/layer_model_layers_74/act/norm":16134.068081992076,"train/train/tensor_act_model_layers_92_post_attention_layernorm/norm":5792.613281256534,"train/train/tensor_act_model_layers_35/mean":-0.015777587890625,"train/train/layer_model_layers_55/act/norm":13753.067646659596,"train/train/tensor_param_model_layers_13_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/std":0.041015625,"train/train/layer_model_layers_39/act/norm":13922.135841342417,"train/train/tensor_grad_model_layers_76_mlp_gate_proj_weight/norm":0.028904190505609408,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/std":4.560229228336443e-05,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/std":0.0322265625,"train/train/tensor_act_model_layers_23_mlp/std":0.03814712066621676,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/max_abs":0.12109375,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/mean":5.979090929031372e-07,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/std":0.02685546875,"train/train/tensor_act_model_layers_72/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79/std":1.708989865447489,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/max_abs":0.27734375,"train/train/layer_model_layers_81/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_q_proj/max_abs":6.09375,"train/train/tensor_param_model_layers_64_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/norm":5.90625,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_40/grad/std":4.7899220572577086e-05,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/mean":-0.00012063980102539062,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_20_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_input_layernorm/std":1.0000003217718978,"train/train/tensor_act_model_layers_90_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/max_abs":0.0017242431640625,"train/train/tensor_act_model_layers_4_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/norm":4.375,"train/train/layer_model_layers_46/grad/std":5.2554970180555764e-05,"train/train/tensor_act_model_layers_53_mlp_down_proj/std":0.07470704115367144,"train/train/tensor_act_model_layers_19_self_attn_q_proj/norm":5451.835039837426,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/std":4.3156770826545536e-05,"train/train/tensor_act_model_layers_89_mlp/mean":-0.00081634521484375,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/std":2.969499420004068e-05,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/std":0.039306640625,"train/train/tensor_act_model_layers_44_mlp/max_abs":0.447265625,"train/train/tensor_act_model_layers_18_self_attn_k_proj/norm":4889.812887775863,"train/train/tensor_act_model_layers_55_mlp_gate_proj/max_abs":2.296875,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/norm":0.0005717816379151319,"train/train/layer__model_layers_7/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp_gate_proj/std":0.2988281560927808,"train/train/global/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/mean":-2.710294211283326e-08,"train/train/tensor_act_model_layers_61_mlp_up_proj/max_abs":2.265625,"train/train/tensor_act_model_layers_85_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/std":9.370964081351324e-05,"train/train/tensor_act_model_layers_83_input_layernorm/max_abs":5.78125,"train/train/tensor_act_model_layers_13_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_v_proj/std":0.41210938252120216,"train/train/tensor_act_model_layers_57_self_attn/max_abs":2.296875,"train/train/tensor_act_model_layers_19_input_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_93/param/norm":26.931662650586578,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/std":0.0458984375,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/std":4.919621314274398e-05,"train/train/tensor_act_model_layers_34_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_89/act/mean":0.011971448148999895,"train/train/tensor_param_model_layers_62_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_24_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/norm":0.032740796171794796,"train/train/tensor_act_model_layers_68_self_attn/std":0.19751904324949665,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/max_abs":0.00018787384033203125,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/mean":-5.324836820363998e-07,"train/train/layer__model_layers_30/param/frac_near_user_limit":0,"train/train/layer_model_layers_4/grad/mean":9.569157839007385e-08,"train/train/tensor_act_model_layers_64_self_attn_v_proj/max_abs":2.4375,"train/train/tensor_grad_model_layers_84_mlp_gate_proj_weight/max_abs":0.00049591064453125,"train/train/tensor_act_model_layers_90/mean":0.0223388671875,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/max_abs":0.00055694580078125,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/max_abs":0.189453125,"train/train/tensor_act_model_layers_61_self_attn_q_proj/std":0.9423843586988069,"train/train/tensor_act_model_layers_77_self_attn_o_proj/mean":-0.0041961669921875,"train/train/tensor_act_model_layers_23_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp_down_proj/max_abs":0.390625,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/max_abs":0.0004062652587890625,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/norm":2.984375,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_post_attention_layernorm/std":1.0000013168892534,"train/train/tensor_act_model_layers_91_self_attn_v_proj/mean":0.0027923583984375,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/max_abs":0.0003528594970703125,"train/train/tensor_act_model_layers_67/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_mlp/std":0.1601562649011605,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_o_proj/max_abs":0.80078125,"train/train/layer_model_layers_62/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/max_abs":0.1416015625,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_1/grad/mean":4.366210221593354e-08,"train/train/tensor_act_model_layers_72/std":1.4961013624541366,"train/train/tensor_act_model_layers_83_input_layernorm/mean":0.006809234619140625,"train/train/tensor_act_model_layers_1_self_attn/max_abs":0.314453125,"train/train/tensor_act_model_layers_49_post_attention_layernorm/std":1.0000003384192833,"train/train/layer_model_layers_32/act/mean":-0.005508405821663993,"train/train/tensor_act_model_layers_47_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/max_abs":0.2216796875,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/std":0.037109375,"train/train/tensor_act_model_layers_88_mlp_up_proj/norm":5009.304561131308,"train/train/layer_model_layers_89/grad/norm":0.066296679458014,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/norm":7.09375,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/std":0.04833984375,"train/train/tensor_act_model_layers_20_mlp_up_proj/norm":2002.2363629956712,"train/train/tensor_grad_model_layers_39_mlp_gate_proj_weight/std":5.034012385261082e-05,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/std":1.6392707893157803e-05,"train/train/tensor_param_model_layers_79_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp/max_abs":0.51953125,"train/train/tensor_act_model_layers_47/max_abs":9.625,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/mean":0.000598907470703125,"train/train/tensor_act_model_layers_17_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_5/param/max_abs":1,"train/train/tensor_act_model_layers_75_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_mlp/std":0.04125977833011162,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/mean":0.00011157989501953125,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/std":0.027099609375,"train/train/tensor_act_model_layers_49_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_down_proj/max_abs":0.427734375,"train/train/tensor_act_model_layers_65_self_attn_v_proj/max_abs":3.078125,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/norm":4.71875,"train/train/tensor_act_model_layers_20_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_self_attn/mean":0.0008687973022460938,"train/train/tensor_param_model_layers_5_mlp_gate_proj_weight/max_abs":0.1005859375,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_mlp_gate_proj_weight/norm":4.96875,"train/train/tensor_act_model_layers_59_self_attn_v_proj/max_abs":2.4375,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8/max_abs":8.875,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/norm":0.002670211790902243,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/mean":-0.000965118408203125,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/std":0.030517578125,"train/train/tensor_grad_model_layers_12_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/std":0.048095703125,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/norm":0.014457442953147921,"train/train/layer_model_layers_59/act/max_abs":9.9375,"train/train/layer_model_layers_11/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp/max_abs":0.37890625,"train/train/tensor_act_model_layers_63_self_attn_o_proj/norm":474.0419062117446,"train/train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_59/act/norm":14619.011556031803,"train/train/tensor_act_model_layers_34_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/max_abs":0.12890625,"train/train/tensor_act_model_layers_35_self_attn_q_proj/std":0.8681657559340238,"train/train/tensor_act_model_layers_90_self_attn_q_proj/max_abs":6.65625,"train/train/tensor_act_model_layers_65_post_attention_layernorm/std":1.0000005115578972,"train/train/tensor_act_model_layers_58_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/max_abs":0.000583648681640625,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/max_abs":0.169921875,"train/train/tensor_param_model_layers_71_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/std":7.452479032897257e-05,"train/train/tensor_act_model_layers_79/max_abs":11.5,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/norm":7.09375,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/norm":0.021853824993428823,"train/train/tensor_act_model_layers_29_mlp_up_proj/max_abs":1.9375,"train/train/tensor_act_model_layers_84_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/max_abs":0.2060546875,"train/train/tensor_act_model_layers_24_mlp_down_proj/mean":0.0008678436279296875,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74/max_abs":11,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/norm":0.03132120603037119,"train/train/tensor_act_model_layers_82_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_44/act/mean":-0.004627738680158343,"train/train/tensor_act_model_layers_39_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_k_proj/mean":0.02862548828125,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/std":3.2157556844396674e-05,"train/train/tensor_act_model_layers_22/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/std":0.0257568359375,"train/train/layer_model_layers_6/grad/max_abs":0.00159454345703125,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/mean":-5.412101745605469e-05,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/max_abs":0.0005340576171875,"train/train/tensor_act_model_layers_91_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp_up_proj/std":0.44531268807401864,"train/train/tensor_act_model_layers_55_input_layernorm/mean":-0.002010345458984375,"train/train/tensor_grad_model_layers_36_mlp_gate_proj_weight/mean":-2.7532223612070084e-08,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp_gate_proj/norm":3889.41413829447,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_mlp_gate_proj_weight/max_abs":0.0004711151123046875,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/norm":0.03031045434872398,"train/train/layer_model_layers_54/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_lm_head/norm":100342.7953681771,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/std":0.0001737438117139026,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/max_abs":0.1708984375,"train/train/tensor_act_model_layers_31_self_attn/max_abs":0.8984375} \ No newline at end of file diff --git a/wandb/run-20260809_050050-59pftr14/logs/debug-core.log b/wandb/run-20260809_050050-59pftr14/logs/debug-core.log new file mode 100644 index 0000000000000000000000000000000000000000..6ab6c95a5b7718c427082e9937cc663568192adb --- /dev/null +++ b/wandb/run-20260809_050050-59pftr14/logs/debug-core.log @@ -0,0 +1,58 @@ +{"time":"2026-08-09T03:56:53.721154509Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmpyqomfhgp/port-2869678.txt","pid":2869678,"detached":false,"idle-timeout":600000000000,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false} +{"time":"2026-08-09T03:56:53.722258921Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":2869678} +{"time":"2026-08-09T03:56:53.722234577Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-2869678-2909401-2521299324/socket","Net":"unix"}} +{"time":"2026-08-09T03:56:53.900147851Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"} +{"time":"2026-08-09T03:58:19.204299995Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"2(@)"} +{"time":"2026-08-09T03:58:19.282614427Z","level":"INFO","msg":"handleInformInit: received","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T03:58:19.544211049Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T03:58:24.897517362Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"xv8tfjtqu29x"} +{"time":"2026-08-09T05:00:38.67636271Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"xv8tfjtqu29x"} +{"time":"2026-08-09T05:00:40.520576799Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"xv8tfjtqu29x"} +{"time":"2026-08-09T05:00:40.552184373Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T05:00:40.553143275Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608126464Z","level":"INFO","msg":"processOutgoingData: finished","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608106419Z","level":"INFO","msg":"connection: closing","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608216462Z","level":"INFO","msg":"connection: closed successfully","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608222262Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"2(@)"} +{"time":"2026-08-09T05:00:50.241711721Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"3(@)"} +{"time":"2026-08-09T05:00:50.316234404Z","level":"INFO","msg":"handleInformInit: received","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:00:50.57530289Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:00:55.905488223Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"4jxnuihlia82"} +{"time":"2026-08-09T05:57:13.765576248Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"4jxnuihlia82"} +{"time":"2026-08-09T05:57:15.666017984Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"4jxnuihlia82"} +{"time":"2026-08-09T05:57:15.999725293Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:57:16.001240777Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068849513Z","level":"INFO","msg":"connection: closing","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068936758Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068853914Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068948948Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"} +{"time":"2026-08-09T05:57:26.287713024Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"4(@)"} +{"time":"2026-08-09T05:57:26.365087395Z","level":"INFO","msg":"handleInformInit: received","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T05:57:26.623707494Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T05:57:31.962348796Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"rz2ldpq16nht"} +{"time":"2026-08-09T07:02:02.153459338Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"rz2ldpq16nht"} +{"time":"2026-08-09T07:02:04.144735717Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"rz2ldpq16nht"} +{"time":"2026-08-09T07:02:04.180963791Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T07:02:04.181861619Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233097548Z","level":"INFO","msg":"connection: closing","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233183816Z","level":"INFO","msg":"connection: closed successfully","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233105065Z","level":"INFO","msg":"processOutgoingData: finished","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233192773Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"4(@)"} +{"time":"2026-08-09T07:02:13.885910885Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"5(@)"} +{"time":"2026-08-09T07:02:13.954769181Z","level":"INFO","msg":"handleInformInit: received","streamId":"uvqyddz0","id":"5(@)"} +{"time":"2026-08-09T07:02:14.215272022Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"uvqyddz0","id":"5(@)"} +{"time":"2026-08-09T07:02:19.530617395Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"fk71ydt1h8f7"} +{"time":"2026-08-09T07:03:39.9336748Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"fk71ydt1h8f7"} +{"time":"2026-08-09T07:03:40.29901782Z","level":"INFO","msg":"connection: closing","id":"5(@)"} +{"time":"2026-08-09T07:03:40.299111085Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"} +{"time":"2026-08-09T07:03:40.299007456Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"} +{"time":"2026-08-09T07:03:40.299122073Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"} +{"time":"2026-08-09T07:03:42.349833633Z","level":"INFO","msg":"connection: closing","id":"1(@)"} +{"time":"2026-08-09T07:03:42.349942779Z","level":"INFO","msg":"connection: closed successfully","id":"1(@)"} +{"time":"2026-08-09T07:03:42.34985531Z","level":"INFO","msg":"processOutgoingData: finished","id":"1(@)"} +{"time":"2026-08-09T07:03:42.349954994Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"1(@)"} +{"time":"2026-08-09T07:03:42.353940558Z","level":"INFO","msg":"server: parent process exited, terminating service process"} +{"time":"2026-08-09T07:03:42.353988499Z","level":"INFO","msg":"server: is shutting down"} +{"time":"2026-08-09T07:03:42.354128892Z","level":"INFO","msg":"server: listener closed","addr":{"Name":"/tmp/wandb-2869678-2909401-2521299324/socket","Net":"unix"}} +{"time":"2026-08-09T07:03:42.354189401Z","level":"INFO","msg":"server: forced shutdown"} +{"time":"2026-08-09T07:03:42.354198506Z","level":"ERROR","msg":"main: Serve() returned error","error":"forced shutdown"} diff --git a/wandb/run-20260809_050050-59pftr14/logs/debug-internal.log b/wandb/run-20260809_050050-59pftr14/logs/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..d90ba4acda959469b020cda295d456100abbbb1a --- /dev/null +++ b/wandb/run-20260809_050050-59pftr14/logs/debug-internal.log @@ -0,0 +1,485 @@ +{"time":"2026-08-09T05:00:50.316382325Z","level":"INFO","msg":"wandb-core"} +{"time":"2026-08-09T05:00:50.316525497Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"} +{"time":"2026-08-09T05:00:50.575130432Z","level":"INFO","msg":"stream: created new stream","id":"59pftr14"} +{"time":"2026-08-09T05:00:50.575204351Z","level":"INFO","msg":"handler: started"} +{"time":"2026-08-09T05:00:50.575297084Z","level":"INFO","msg":"stream: started"} +{"time":"2026-08-09T05:00:50.575306959Z","level":"INFO","msg":"writer: started","stream_id":"59pftr14"} +{"time":"2026-08-09T05:00:50.575368938Z","level":"INFO","msg":"sender: started"} +{"time":"2026-08-09T05:00:51.480743748Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1} +{"time":"2026-08-09T05:00:51.561240291Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:01:06.481345427Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":0,"events_lines":2,"console_offset":0,"console_lines":5,"uploaded_len":2} +{"time":"2026-08-09T05:01:06.588785368Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:01:21.481259493Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":2,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:01:21.590035378Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:01:32.681908293Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":384} +{"time":"2026-08-09T05:01:32.68198076Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1} +{"time":"2026-08-09T05:01:32.691539462Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":2462} +{"time":"2026-08-09T05:01:32.692719035Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":335} +{"time":"2026-08-09T05:01:32.698314928Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":4822} +{"time":"2026-08-09T05:01:32.698424691Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":14} +{"time":"2026-08-09T05:01:32.702256488Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":6358} +{"time":"2026-08-09T05:01:32.702387381Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":15} +{"time":"2026-08-09T05:01:32.706713558Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":7253} +{"time":"2026-08-09T05:01:32.708196871Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":93} +{"time":"2026-08-09T05:01:32.712656212Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":8892} +{"time":"2026-08-09T05:01:32.725765023Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":5668} +{"time":"2026-08-09T05:01:32.727393292Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":14924} +{"time":"2026-08-09T05:01:32.727423186Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1} +{"time":"2026-08-09T05:01:32.727732623Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":15045} +{"time":"2026-08-09T05:01:32.727835605Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":9} +{"time":"2026-08-09T05:01:32.730442968Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":16070} +{"time":"2026-08-09T05:01:32.737046002Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":2678} +{"time":"2026-08-09T05:01:36.519407838Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":1,"events_offset":4,"events_lines":2,"console_offset":4,"console_lines":2} +{"time":"2026-08-09T05:01:37.706004293Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:01:51.481365298Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":6,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:01:51.596320975Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:02:06.480925245Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":8,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:02:06.586636942Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:02:21.498041897Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":1,"history_lines":1,"events_offset":10,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:02:22.532929656Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:02:36.481659249Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":12,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:02:36.621746818Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:02:51.507187931Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":2,"history_lines":1,"events_offset":14,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:02:52.538965872Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:03:06.497711137Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":3,"history_lines":1,"events_offset":16,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:03:07.406445161Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:03:21.481660818Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":18,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:03:21.644645315Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:03:36.481585895Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":20,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:03:36.591305248Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:03:51.499410115Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":4,"history_lines":1,"events_offset":22,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:03:52.420339049Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:04:06.481223309Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":24,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:04:06.586916017Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:04:21.499976608Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":5,"history_lines":1,"events_offset":26,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:04:22.440880355Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:04:36.503115163Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":6,"history_lines":1,"events_offset":28,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:04:37.434101869Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:04:51.480868859Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":30,"events_lines":2,"console_offset":6,"console_lines":11} +{"time":"2026-08-09T05:04:51.58771127Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:05:06.48149862Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":32,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:05:06.617849821Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:05:21.498001668Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":7,"history_lines":1,"events_offset":34,"events_lines":2,"console_offset":16,"console_lines":2} +{"time":"2026-08-09T05:05:22.532059405Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:05:36.481272157Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":36,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:05:36.611686864Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:05:51.50349838Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":8,"history_lines":1,"events_offset":38,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:05:52.42872915Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:06:06.481067013Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":40,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:06:06.624376428Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:06:21.481451995Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":42,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:06:21.580400913Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:06:36.498750788Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":9,"history_lines":1,"events_offset":44,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:06:37.479364541Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:06:51.499950008Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":10,"history_lines":1,"events_offset":46,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:06:52.424905167Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:07:06.480837832Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":48,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:07:06.5879285Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:07:21.497769405Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":11,"history_lines":1,"events_offset":50,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:07:22.420651885Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:07:36.481679224Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":52,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:07:36.586207773Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:07:51.48168829Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":54,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:07:51.608316231Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:08:06.497385296Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":12,"history_lines":1,"events_offset":56,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:08:07.414492478Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:08:21.498696248Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":13,"history_lines":1,"events_offset":58,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:08:22.472074441Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:08:36.481080877Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":60,"events_lines":2,"console_offset":18,"console_lines":11} +{"time":"2026-08-09T05:08:36.593043174Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:08:51.481213802Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":62,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T05:08:51.601597348Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:09:06.498299771Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":14,"history_lines":1,"events_offset":64,"events_lines":2,"console_offset":28,"console_lines":2} +{"time":"2026-08-09T05:09:07.594657221Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:09:21.480997622Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":66,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T05:09:21.587300995Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:09:36.500970166Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":15,"history_lines":1,"events_offset":68,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T05:09:37.412755991Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:09:51.481568492Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":70,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T05:09:51.587311446Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:10:06.503035144Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":16,"history_lines":1,"events_offset":72,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T05:10:07.401707034Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:10:21.481629645Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":74,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T05:10:21.652792794Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:10:36.504128018Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":17,"history_lines":1,"events_offset":76,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T05:10:37.451086196Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:10:51.481374598Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":78,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T05:10:51.616703256Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:11:06.498458682Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":18,"history_lines":1,"events_offset":80,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T05:11:07.69317861Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:11:21.481191292Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":82,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T05:11:21.58886328Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:11:36.481561709Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":84,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T05:11:36.583857842Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:11:51.497386796Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":19,"history_lines":1,"events_offset":86,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T05:11:52.440204013Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:12:06.497327915Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":20,"history_lines":1,"events_offset":88,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T05:12:07.396293595Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:12:21.481158326Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":90,"events_lines":2,"console_offset":30,"console_lines":11} +{"time":"2026-08-09T05:12:21.592522674Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:12:36.503425179Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":21,"history_lines":1,"events_offset":92,"events_lines":2,"console_offset":40,"console_lines":2} +{"time":"2026-08-09T05:12:37.416814559Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:12:51.481041129Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":94,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T05:12:51.610448799Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:13:06.481201194Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":96,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T05:13:06.625500113Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:13:21.498075795Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":22,"history_lines":1,"events_offset":98,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T05:13:22.560395397Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:13:36.481411011Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":100,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T05:13:36.605888576Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:13:51.500833721Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":23,"history_lines":1,"events_offset":102,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T05:13:52.407787379Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:14:06.497449559Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":24,"history_lines":1,"events_offset":104,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T05:14:07.390239326Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:14:21.48119166Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":106,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T05:14:21.638691331Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:14:36.481262857Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":108,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T05:14:36.634845955Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:14:51.501864135Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":25,"history_lines":1,"events_offset":110,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T05:14:52.366198559Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:15:06.481264357Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":112,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T05:15:06.577661402Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:15:21.498953568Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":26,"history_lines":1,"events_offset":114,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T05:15:22.437573941Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:15:36.501716856Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":27,"history_lines":1,"events_offset":116,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T05:15:37.369161631Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:15:51.481510049Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":118,"events_lines":2,"console_offset":42,"console_lines":11} +{"time":"2026-08-09T05:15:51.63079258Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:16:06.481670382Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":120,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T05:16:06.612349198Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:16:21.497081864Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":28,"history_lines":1,"events_offset":122,"events_lines":2,"console_offset":52,"console_lines":2} +{"time":"2026-08-09T05:16:22.429585439Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:16:36.481386631Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":124,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T05:16:36.716895363Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:16:51.498344862Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":29,"history_lines":1,"events_offset":126,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T05:16:52.573229517Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:17:06.481679973Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":128,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T05:17:06.6380613Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:17:21.481192289Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":130,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T05:17:21.615467345Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:17:36.498032747Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":30,"history_lines":1,"events_offset":132,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T05:17:37.37259259Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:17:51.501849113Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":31,"history_lines":1,"events_offset":134,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T05:17:52.461736682Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:18:06.481331592Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":136,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T05:18:06.664869101Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:18:21.50046419Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":32,"history_lines":1,"events_offset":138,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T05:18:22.389205003Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:18:36.481352846Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":140,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T05:18:36.69619412Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:18:51.48085544Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":142,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T05:18:51.599415662Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:19:06.501115812Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":33,"history_lines":1,"events_offset":144,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T05:19:07.486241738Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:19:21.499147294Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":34,"history_lines":1,"events_offset":146,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T05:19:22.438925205Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:19:36.481577527Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":148,"events_lines":2,"console_offset":54,"console_lines":13} +{"time":"2026-08-09T05:19:36.608439891Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:19:51.481485963Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":150,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T05:19:51.659884173Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:20:06.525040764Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":35,"history_lines":1,"events_offset":152,"events_lines":2,"console_offset":66,"console_lines":2} +{"time":"2026-08-09T05:20:07.675864743Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:20:21.481552724Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":154,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T05:20:21.581914998Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:20:36.500684642Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":36,"history_lines":1,"events_offset":156,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T05:20:37.523239345Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:20:51.481557042Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":158,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T05:20:51.618146016Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:21:06.481365631Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":160,"events_lines":2,"console_offset":68,"console_lines":2} +{"time":"2026-08-09T05:21:06.603391244Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:21:21.505549859Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":37,"history_lines":1,"events_offset":162,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T05:21:22.433779828Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:21:36.502373763Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":38,"history_lines":1,"events_offset":164,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T05:21:37.46846612Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:21:51.481185661Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":166,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T05:21:51.678879589Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:22:06.504549408Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":39,"history_lines":1,"events_offset":168,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T05:22:07.431785778Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:22:21.480925067Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":170,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T05:22:21.617766324Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:22:36.481037106Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":172,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T05:22:36.631472656Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:22:51.498329537Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":40,"history_lines":1,"events_offset":174,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T05:22:52.376455957Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:23:06.502499439Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":41,"history_lines":1,"events_offset":176,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T05:23:07.406128001Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:23:21.481598535Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":178,"events_lines":2,"console_offset":69,"console_lines":10} +{"time":"2026-08-09T05:23:21.621966825Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:23:36.498090022Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":42,"history_lines":1,"events_offset":180,"events_lines":2,"console_offset":78,"console_lines":2} +{"time":"2026-08-09T05:23:37.601595059Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:23:51.481759024Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":182,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T05:23:51.586781813Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:24:06.481276767Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":184,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T05:24:06.598097548Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:24:21.507460244Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":43,"history_lines":1,"events_offset":186,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T05:24:22.440295279Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:24:36.481283213Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":188,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T05:24:36.610184037Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:24:51.502397715Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":44,"history_lines":1,"events_offset":190,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T05:24:52.438748096Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:25:06.501113438Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":45,"history_lines":1,"events_offset":192,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T05:25:07.42127244Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:25:21.481510164Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":194,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T05:25:21.669561341Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:25:36.481578249Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":196,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T05:25:36.594439235Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:25:51.498405234Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":46,"history_lines":1,"events_offset":198,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T05:25:52.378590133Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:26:06.480986539Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":200,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T05:26:06.638024923Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:26:21.50308839Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":47,"history_lines":1,"events_offset":202,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T05:26:22.404563682Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:26:36.48164935Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":204,"events_lines":2,"console_offset":80,"console_lines":6} +{"time":"2026-08-09T05:26:36.608214624Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:26:51.498904739Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":48,"history_lines":1,"events_offset":206,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T05:26:52.54906005Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:27:06.481195872Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":208,"events_lines":2,"console_offset":81,"console_lines":1} +{"time":"2026-08-09T05:27:06.694166718Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:27:21.504577552Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":49,"history_lines":1,"events_offset":210,"events_lines":2,"console_offset":86,"console_lines":6} +{"time":"2026-08-09T05:27:22.424460773Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:27:36.481110203Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":212,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T05:27:36.604006133Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:27:51.481765731Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":214,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T05:27:51.611864716Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:28:06.498158704Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":50,"history_lines":1,"events_offset":216,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T05:28:07.409376378Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:28:21.481511802Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":218,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T05:28:21.562499861Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:28:36.50710607Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":51,"history_lines":1,"events_offset":220,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T05:28:37.487154585Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:28:51.497901326Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":52,"history_lines":1,"events_offset":222,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T05:28:52.446794195Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:29:06.481006896Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":224,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T05:29:06.604939782Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:29:21.48132648Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":226,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T05:29:21.675663702Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:29:36.502055175Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":53,"history_lines":1,"events_offset":228,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T05:29:37.492772989Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:29:51.48098667Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":230,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T05:29:51.566571956Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:30:06.498304058Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":54,"history_lines":1,"events_offset":232,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T05:30:07.476602804Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:30:21.497009392Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":55,"history_lines":1,"events_offset":234,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T05:30:22.426875321Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:30:36.481198656Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":236,"events_lines":2,"console_offset":92,"console_lines":11} +{"time":"2026-08-09T05:30:36.569501322Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:30:51.481362955Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":238,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T05:30:51.58612004Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:31:06.498109113Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":56,"history_lines":1,"events_offset":240,"events_lines":2,"console_offset":102,"console_lines":2} +{"time":"2026-08-09T05:31:07.465651065Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:31:21.481204233Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":242,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T05:31:21.607697687Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:31:36.501843353Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":57,"history_lines":1,"events_offset":244,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T05:31:37.484272619Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:31:51.481294572Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":246,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T05:31:51.569699481Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:32:06.481267101Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":248,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T05:32:06.564697645Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:32:21.501309751Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":58,"history_lines":1,"events_offset":250,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T05:32:22.400233409Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:32:36.502665345Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":59,"history_lines":1,"events_offset":252,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T05:32:37.539187444Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:32:51.481000342Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":254,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T05:32:51.589307243Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:33:06.497296324Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":60,"history_lines":1,"events_offset":256,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T05:33:07.410411774Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:33:21.480896144Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":258,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T05:33:21.557619437Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:33:36.481359105Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":260,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T05:33:36.586477785Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:33:51.501293775Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":61,"history_lines":1,"events_offset":262,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T05:33:52.446357727Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:34:06.501566811Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":62,"history_lines":1,"events_offset":264,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T05:34:07.431632319Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:34:21.481356503Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":266,"events_lines":2,"console_offset":104,"console_lines":11} +{"time":"2026-08-09T05:34:21.58737243Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:34:36.481254313Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":268,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T05:34:36.596785368Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:34:51.498620247Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":63,"history_lines":1,"events_offset":270,"events_lines":2,"console_offset":114,"console_lines":2} +{"time":"2026-08-09T05:34:52.447346801Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:35:06.481384808Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":272,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T05:35:06.584314432Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:35:21.499491427Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":64,"history_lines":1,"events_offset":274,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T05:35:22.406403243Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:35:36.481510694Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":276,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T05:35:36.570794108Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:35:51.502728659Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":65,"history_lines":1,"events_offset":278,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T05:35:52.433176071Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:36:06.481082158Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":280,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T05:36:06.56595802Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:36:21.502575907Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":66,"history_lines":1,"events_offset":282,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T05:36:22.396774635Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:36:36.481194169Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":284,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T05:36:36.564324046Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:36:51.507585877Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":67,"history_lines":1,"events_offset":286,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T05:36:52.450052493Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:37:06.481177189Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":288,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T05:37:06.562873382Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:37:21.481773468Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":290,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T05:37:21.578202075Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:37:36.496352178Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":68,"history_lines":1,"events_offset":292,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T05:37:37.378367849Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:37:51.498182421Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":69,"history_lines":1,"events_offset":294,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T05:37:52.501913241Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:38:06.481276719Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":296,"events_lines":2,"console_offset":116,"console_lines":13} +{"time":"2026-08-09T05:38:06.575845243Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:38:21.517547479Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":70,"history_lines":1,"events_offset":298,"events_lines":2,"console_offset":128,"console_lines":2} +{"time":"2026-08-09T05:38:22.68375147Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:38:36.48169975Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":300,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T05:38:36.58233458Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:38:51.481639785Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":302,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T05:38:51.57204535Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:39:06.502117161Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":71,"history_lines":1,"events_offset":304,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T05:39:07.408069696Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:39:21.481127559Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":306,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T05:39:21.584984533Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:39:36.500833886Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":72,"history_lines":1,"events_offset":308,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T05:39:37.3542055Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:39:51.499790298Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":73,"history_lines":1,"events_offset":310,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T05:39:52.423275386Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:40:06.481402259Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":312,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T05:40:06.574340065Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:40:21.481696586Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":314,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T05:40:21.585374046Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:40:36.500319073Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":74,"history_lines":1,"events_offset":316,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T05:40:37.416603366Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:40:51.481735415Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":318,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T05:40:51.568608859Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:41:06.481327101Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":320,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T05:41:06.577508114Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:41:21.504914431Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":75,"history_lines":1,"events_offset":322,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T05:41:22.589623821Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:41:36.500920471Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":76,"history_lines":1,"events_offset":324,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T05:41:37.445500069Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:41:51.481699831Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":326,"events_lines":2,"console_offset":130,"console_lines":11} +{"time":"2026-08-09T05:41:51.589802594Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:42:06.499496996Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":77,"history_lines":1,"events_offset":328,"events_lines":2,"console_offset":140,"console_lines":2} +{"time":"2026-08-09T05:42:07.490323728Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:42:21.481007487Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":330,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T05:42:21.573910407Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:42:36.480944829Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":332,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T05:42:36.58648518Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:42:51.502651759Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":78,"history_lines":1,"events_offset":334,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T05:42:52.407092114Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:43:06.48131406Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":336,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T05:43:06.572964826Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:43:21.502659005Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":79,"history_lines":1,"events_offset":338,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T05:43:22.460952902Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:43:36.498438868Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":80,"history_lines":1,"events_offset":340,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T05:43:37.435539856Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:43:51.481499575Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":342,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T05:43:51.581657982Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:44:06.480992155Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":344,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T05:44:06.575942046Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:44:21.499285771Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":81,"history_lines":1,"events_offset":346,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T05:44:22.48502359Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:44:36.481415886Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":348,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T05:44:36.590027025Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:44:51.498199637Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":82,"history_lines":1,"events_offset":350,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T05:44:52.522574549Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:45:06.507362609Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":83,"history_lines":1,"events_offset":352,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T05:45:07.437618138Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:45:21.481223976Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":354,"events_lines":2,"console_offset":142,"console_lines":11} +{"time":"2026-08-09T05:45:21.571863157Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:45:36.481250148Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":356,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T05:45:36.588306221Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:45:51.498140841Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":84,"history_lines":1,"events_offset":358,"events_lines":2,"console_offset":152,"console_lines":2} +{"time":"2026-08-09T05:45:52.40392286Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:46:06.481151661Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":360,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T05:46:06.574924044Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:46:21.50224276Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":85,"history_lines":1,"events_offset":362,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T05:46:22.468575161Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:46:36.48151792Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":364,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T05:46:36.557472174Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:46:51.481659212Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":366,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T05:46:51.583561336Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:47:06.503251857Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":86,"history_lines":1,"events_offset":368,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T05:47:07.524513415Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:47:21.498223553Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":87,"history_lines":1,"events_offset":370,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T05:47:22.436952446Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:47:36.481235917Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":372,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T05:47:36.58953031Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:47:51.509767914Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":88,"history_lines":1,"events_offset":374,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T05:47:52.472129123Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:48:06.480977725Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":376,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T05:48:06.572015048Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:48:21.481635627Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":378,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T05:48:21.572405344Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:48:36.522493716Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":89,"history_lines":1,"events_offset":380,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T05:48:37.446677623Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:48:51.501843279Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":90,"history_lines":1,"events_offset":382,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T05:48:52.491223677Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:49:06.481489796Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":384,"events_lines":2,"console_offset":154,"console_lines":11} +{"time":"2026-08-09T05:49:06.586343176Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:49:21.499852391Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":91,"history_lines":1,"events_offset":386,"events_lines":2,"console_offset":164,"console_lines":2} +{"time":"2026-08-09T05:49:22.479462348Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:49:36.481384045Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":388,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T05:49:36.5774241Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:49:51.481275624Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":390,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T05:49:51.585688059Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:50:06.501668608Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":92,"history_lines":1,"events_offset":392,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T05:50:07.367218244Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:50:21.481792859Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":394,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T05:50:21.570177867Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:50:36.481176295Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":396,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T05:50:36.58018685Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:50:51.496247888Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":93,"history_lines":1,"events_offset":398,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T05:50:52.481878699Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:51:06.497846529Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":94,"history_lines":1,"events_offset":400,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T05:51:07.407970871Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:51:21.481238939Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":402,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T05:51:21.59538021Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:51:36.481689771Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":404,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T05:51:36.583888684Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:51:51.499980191Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":95,"history_lines":1,"events_offset":406,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T05:51:52.427907364Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:52:06.481473609Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":408,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T05:52:06.597914659Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:52:21.499728804Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":96,"history_lines":1,"events_offset":410,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T05:52:22.366340564Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:52:36.498067451Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":97,"history_lines":1,"events_offset":412,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T05:52:37.449598672Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:52:51.481617633Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":414,"events_lines":2,"console_offset":166,"console_lines":11} +{"time":"2026-08-09T05:52:51.581562046Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:53:06.481233846Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":416,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:53:06.57963203Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:53:21.499870228Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":98,"history_lines":1,"events_offset":418,"events_lines":2,"console_offset":176,"console_lines":2} +{"time":"2026-08-09T05:53:22.41340383Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:53:36.481163388Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":420,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:53:36.57497954Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:53:51.497905926Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":99,"history_lines":1,"events_offset":422,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:53:52.368218191Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:54:06.481748072Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":424,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:54:06.578247243Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:54:21.481590391Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":426,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:54:21.595103309Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:54:36.504538021Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":100,"history_lines":1,"events_offset":428,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:54:37.426563022Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:54:51.480970743Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":430,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:54:51.573296853Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:55:06.505576648Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":101,"history_lines":1,"events_offset":432,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:55:07.439584344Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:55:21.481626327Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":434,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:55:21.573957985Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:55:36.481205668Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":436,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:55:36.566940604Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:55:51.502848806Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":102,"history_lines":1,"events_offset":438,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:55:52.39689707Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:56:06.481314164Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":440,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:56:06.569483549Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:56:21.481079544Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":442,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:56:21.588552813Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:56:36.497668927Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":103,"history_lines":1,"events_offset":444,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:56:37.466258007Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:56:51.502614572Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":104,"history_lines":2,"events_offset":446,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T05:56:52.423617477Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:57:06.481096852Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":448,"events_lines":2,"console_offset":178,"console_lines":13} +{"time":"2026-08-09T05:57:06.567967592Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:57:14.637598114Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"} +{"time":"2026-08-09T05:57:14.659564926Z","level":"INFO","msg":"filestream: sending request","total_files":3,"history_offset":106,"history_lines":1,"console_offset":190,"console_lines":31,"uploaded_len":3,"complete":true,"exit_code":0} +{"time":"2026-08-09T05:57:15.639628923Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:57:15.641071554Z","level":"INFO","msg":"handler: operation stats","stats":{}} +{"time":"2026-08-09T05:57:15.999773399Z","level":"INFO","msg":"stream: finishing up"} +{"time":"2026-08-09T05:57:15.999820775Z","level":"INFO","msg":"handler: closed"} +{"time":"2026-08-09T05:57:15.999975766Z","level":"INFO","msg":"sender: closed"} +{"time":"2026-08-09T05:57:15.999980748Z","level":"INFO","msg":"stream: all finished"} diff --git a/wandb/run-20260809_050050-59pftr14/logs/debug.log b/wandb/run-20260809_050050-59pftr14/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..ea2b2b26e975f7f88f0cf903666ee00e6e2d72ee --- /dev/null +++ b/wandb/run-20260809_050050-59pftr14/logs/debug.log @@ -0,0 +1,28 @@ +2026-08-09 05:00:50,314 INFO MainThread:3518447 [wandb_setup.py:_flush():81] Current SDK version is 0.28.1 +2026-08-09 05:00:50,314 INFO MainThread:3518447 [wandb_setup.py:_flush():81] Configure stats pid to 3518447 +2026-08-09 05:00:50,314 INFO MainThread:3518447 [wandb_setup.py:_flush():81] Loading settings from environment variables +2026-08-09 05:00:50,314 INFO MainThread:3518447 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_050050-59pftr14/logs/debug.log +2026-08-09 05:00:50,314 INFO MainThread:3518447 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_050050-59pftr14/logs/debug-internal.log +2026-08-09 05:00:50,315 INFO MainThread:3518447 [wandb_init.py:init():772] calling init triggers +2026-08-09 05:00:50,315 INFO MainThread:3518447 [wandb_init.py:init():777] wandb.init called with sweep_config: {} +config: {'_wandb': {}} +2026-08-09 05:00:50,315 INFO MainThread:3518447 [wandb_init.py:init():820] starting backend +2026-08-09 05:00:50,315 INFO MainThread:3518447 [wandb_init.py:init():826] Connected to an existing wandb-core service via WANDB_SERVICE +2026-08-09 05:00:50,315 INFO MainThread:3518447 [wandb_init.py:init():835] sending inform_init request +2026-08-09 05:00:50,575 INFO MainThread:3518447 [wandb_init.py:init():840] backend started and connected +2026-08-09 05:00:50,579 INFO MainThread:3518447 [wandb_init.py:init():910] updated telemetry +2026-08-09 05:00:50,585 INFO MainThread:3518447 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout +2026-08-09 05:00:50,829 INFO MainThread:3518447 [wandb_init.py:init():978] starting run threads in backend +2026-08-09 05:00:50,902 INFO MainThread:3518447 [wandb_run.py:_console_start():2621] atexit reg +2026-08-09 05:00:50,902 INFO MainThread:3518447 [wandb_run.py:_redirect():2471] redirect: wrap_raw +2026-08-09 05:00:50,902 INFO MainThread:3518447 [wandb_run.py:_redirect():2540] Wrapping output streams. +2026-08-09 05:00:50,903 INFO MainThread:3518447 [wandb_run.py:_redirect():2563] Redirects installed. +2026-08-09 05:00:50,905 INFO MainThread:3518447 [wandb_init.py:init():1016] run started, returning control to user process +2026-08-09 05:00:50,906 INFO MainThread:3518447 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.15.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 94, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'glu', 'activation': 'linear', 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/glu-linear-94L_run', 'per_device_train_batch_size': 128, 'num_train_epochs': 1, 'max_steps': 1500, 'learning_rate': 0.001, 'lr_scheduler_type': 'constant', 'lr_scheduler_kwargs': None, 'warmup_steps': 0, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.01, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 4, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-glu-linear-94L-15.9M-20260809-050049', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 50, 'eval_delay': 0, 'per_device_eval_batch_size': 128, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': True, 'hub_token': '', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/A-glu-linear-94L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1} +2026-08-09 05:00:50,910 INFO MainThread:3518447 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 15949440 - > +2026-08-09 05:00:50,910 INFO MainThread:3518447 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 15949440 None +2026-08-09 05:57:13,764 INFO MainThread:3518447 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/huggingface/59pftr14 +2026-08-09 05:57:13,765 INFO MainThread:3518447 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0 +2026-08-09 05:57:13,765 INFO MainThread:3518447 [wandb_run.py:_restore():2570] restore +2026-08-09 05:57:13,765 INFO MainThread:3518447 [wandb_run.py:_restore():2576] restore done +2026-08-09 05:57:15,998 INFO MainThread:3518447 [wandb_run.py:_footer_sync_info():3993] logging synced files diff --git a/wandb/run-20260809_050050-59pftr14/run-59pftr14.wandb b/wandb/run-20260809_050050-59pftr14/run-59pftr14.wandb new file mode 100644 index 0000000000000000000000000000000000000000..73533bee24bbc954e4cd723f6762f4c4a5d5c24d --- /dev/null +++ b/wandb/run-20260809_050050-59pftr14/run-59pftr14.wandb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e15846733ea8053359a05fee6d71eff0f773e49c1dc69509b91f5d7618e26991 +size 18403101 diff --git a/wandb/run-20260809_055134-ppvwto7l/files/config.yaml b/wandb/run-20260809_055134-ppvwto7l/files/config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7eb3bf2651bf39f67832c686a34acea064d96d86 --- /dev/null +++ b/wandb/run-20260809_055134-ppvwto7l/files/config.yaml @@ -0,0 +1,431 @@ +_name_or_path: + value: "" +_wandb: + value: + cli_version: 0.28.1 + e: + 9oiz0iyxjogu9rr8lhkve9eqryk4v0ll: + args: + - --config + - /mnt/data/zainulabideen/zain-exp/notebooks/Activation/configs/baseline1.yaml + - --variants + - mlp-linear-9L + - --push + codePath: sweep.py + codePathLocal: sweep.py + cpu_count: 112 + cpu_count_logical: 224 + cudaVersion: "12.4" + disk: + /: + total: "1560765693952" + used: "708272566272" + email: deepnevro@gmail.com + executable: /mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python + git: + commit: 34b8d2e8f9a0c5751333310e69fa0c1056381deb + remote: https://github.com/deepnevro/Activation.git + gpu: NVIDIA H100 80GB HBM3 + gpu_count: 8 + gpu_nvidia: + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-39c684a5-fde6-83d7-1663-0859795881ae + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-132944c4-b689-2b5f-89a4-d730401677ab + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bc6c3e3c-9b90-09ca-c034-774961847c54 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea + host: deeplens-k3s-node1 + memory: + total: "2164089937920" + os: Linux-5.15.0-126-generic-x86_64-with-glibc2.35 + program: /mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py + python: CPython 3.11.15 + root: /mnt/data/zainulabideen/zain-exp/notebooks/Activation + startedAt: "2026-08-09T05:51:34.110899Z" + writerId: 9oiz0iyxjogu9rr8lhkve9eqryk4v0ll + m: + - "1": train/global_step + "6": + - 3 + "7": [] + - "2": '*' + "5": 1 + "6": + - 1 + "7": [] + python_version: 3.11.15 + t: + "1": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "2": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "3": + - 2 + - 7 + - 13 + - 19 + - 66 + "4": 3.11.15 + "5": 0.28.1 + "6": 5.15.0.dev0 + "9": + "1": transformers_trainer + "12": 0.28.1 + "13": linux-x86_64 +accelerator_config: + value: + dispatch_batches: null + even_batches: true + gradient_accumulation_kwargs: null + non_blocking: false + split_batches: false + use_seedable_sampler: true +activation: + value: linear +adam_beta1: + value: 0.9 +adam_beta2: + value: 0.999 +adam_epsilon: + value: 1e-08 +architectures: + value: null +attention_bias: + value: false +attention_dropout: + value: 0 +auto_find_batch_size: + value: false +average_tokens_across_devices: + value: true +batch_eval_metrics: + value: false +bf16: + value: true +bf16_full_eval: + value: false +bos_token_id: + value: 1 +chunk_size_feed_forward: + value: 0 +data_seed: + value: 42 +dataloader_drop_last: + value: false +dataloader_in_order: + value: true +dataloader_multiprocessing_context: + value: null +dataloader_num_workers: + value: 0 +dataloader_persistent_workers: + value: false +dataloader_pin_memory: + value: true +dataloader_prefetch_factor: + value: null +ddp_backend: + value: null +ddp_broadcast_buffers: + value: null +ddp_bucket_cap_mb: + value: null +ddp_find_unused_parameters: + value: null +ddp_static_graph: + value: null +ddp_timeout: + value: 1800 +debug: + value: [] +deepspeed: + value: null +disable_tqdm: + value: false +do_eval: + value: true +do_predict: + value: false +do_train: + value: false +dtype: + value: null +enable_jit_checkpoint: + value: false +eos_token_id: + value: 2 +eval_accumulation_steps: + value: null +eval_delay: + value: 0 +eval_do_concat_batches: + value: true +eval_on_start: + value: false +eval_steps: + value: 50 +eval_strategy: + value: steps +eval_use_gather_object: + value: false +fp16: + value: false +fp16_full_eval: + value: false +fsdp: + value: null +fsdp_config: + value: null +full_determinism: + value: false +gradient_accumulation_steps: + value: 16 +gradient_checkpointing: + value: false +gradient_checkpointing_kwargs: + value: null +greater_is_better: + value: null +head_dim: + value: 32 +hidden_act: + value: silu +hidden_size: + value: 128 +hub_always_push: + value: false +hub_model_id: + value: w-ahmad/finale-mlp-linear-9L +hub_private_repo: + value: null +hub_revision: + value: null +hub_strategy: + value: every_save +hub_token: + value: +id2label: + value: + "0": LABEL_0 + "1": LABEL_1 +ignore_data_skip: + value: false +include_for_metrics: + value: [] +include_num_input_tokens_seen: + value: "no" +initializer_range: + value: 0.02 +intermediate_size: + value: 256 +is_encoder_decoder: + value: false +label_names: + value: null +label_smoothing_factor: + value: 0 +label2id: + value: + LABEL_0: 0 + LABEL_1: 1 +learning_rate: + value: 0.0005 +length_column_name: + value: length +liger_kernel_config: + value: null +load_best_model_at_end: + value: false +local_rank: + value: -1 +log_level: + value: passive +log_level_replica: + value: warning +log_on_each_node: + value: true +logging_first_step: + value: false +logging_nan_inf_filter: + value: true +logging_steps: + value: 20 +logging_strategy: + value: steps +lr_scheduler_kwargs: + value: null +lr_scheduler_type: + value: constant +max_grad_norm: + value: 1 +max_position_embeddings: + value: 512 +max_steps: + value: 750 +metric_for_best_model: + value: null +mlp_bias: + value: false +mlp_type: + value: mlp +model/num_parameters: + value: 2001280 +model_type: + value: tiny_llama +neftune_noise_alpha: + value: null +num_attention_heads: + value: 4 +num_hidden_layers: + value: 9 +num_key_value_heads: + value: 4 +num_train_epochs: + value: 1 +optim: + value: adamw_torch_fused +optim_args: + value: null +optim_target_modules: + value: null +output_attentions: + value: false +output_dir: + value: outio/mlp-linear-9L_run +output_hidden_states: + value: false +pad_token_id: + value: 0 +parallelism_config: + value: null +per_device_eval_batch_size: + value: 80 +per_device_train_batch_size: + value: 80 +prediction_loss_only: + value: false +pretraining_tp: + value: 1 +problem_type: + value: null +project: + value: huggingface +push_to_hub: + value: true +remove_unused_columns: + value: false +report_to: + value: + - wandb +restore_callback_states_from_checkpoint: + value: false +resume_from_checkpoint: + value: null +return_dict: + value: true +rms_norm_eps: + value: 1e-06 +rope_parameters: + value: + rope_theta: 10000 + rope_type: default +run_name: + value: LM-mlp-linear-9L-2.0M-20260809-055132 +save_on_each_node: + value: false +save_only_model: + value: false +save_steps: + value: 100 +save_strategy: + value: steps +save_total_limit: + value: null +seed: + value: 42 +skip_memory_metrics: + value: true +tf32: + value: null +tie_word_embeddings: + value: true +tokenizer_name: + value: w-ahmad/tiny-stories-tokenizer +torch_compile: + value: false +torch_compile_backend: + value: null +torch_compile_mode: + value: null +torch_empty_cache_steps: + value: null +trackio_bucket_id: + value: null +trackio_space_id: + value: null +trackio_static_space_id: + value: null +train_sampling_strategy: + value: random +transformers_version: + value: 5.15.0.dev0 +use_cache: + value: false +use_cpu: + value: false +use_liger_kernel: + value: false +vocab_size: + value: 4096 +warmup_steps: + value: 0 +weight_decay: + value: 0.01 diff --git a/wandb/run-20260809_055134-ppvwto7l/files/requirements.txt b/wandb/run-20260809_055134-ppvwto7l/files/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..123b15ebf857859624f7f4332e92341c8ef13fdf --- /dev/null +++ b/wandb/run-20260809_055134-ppvwto7l/files/requirements.txt @@ -0,0 +1,149 @@ +asttokens==3.0.1 +comm==0.2.3 +debugpy==1.8.21 +decorator==5.3.1 +executing==2.2.1 +nest-asyncio==1.6.0 +parso==0.8.7 +platformdirs==4.11.0 +psutil==7.2.2 +ptyprocess==0.7.0 +pure_eval==0.2.3 +Pygments==2.20.0 +pyzmq==27.1.0 +setuptools==83.0.0 +six==1.17.0 +tornado==6.5.7 +traitlets==5.15.0 +fsspec==2026.4.0 +wcwidth==0.8.2 +ipython_pygments_lexers==1.1.1 +jedi==0.20.0 +jupyter_core==5.9.1 +matplotlib-inline==0.2.2 +pexpect==4.9.0 +prompt_toolkit==3.0.53 +python-dateutil==2.9.0.post0 +stack_data==0.6.3 +wheel==0.47.0 +jupyter_client==8.9.1 +pip==26.1.2 +ipython==9.15.0 +ipykernel==7.2.0 +threadpoolctl==3.6.0 +pyparsing==3.3.2 +typing_extensions==4.15.0 +Jinja2==3.1.6 +narwhals==2.24.0 +kiwisolver==1.5.0 +joblib==1.5.3 +fonttools==4.63.0 +cycler==0.12.1 +scipy==1.17.1 +pandas==3.0.5 +contourpy==1.3.3 +scikit-learn==1.9.0 +matplotlib==3.11.1 +urllib3==2.7.0 +tqdm==4.70.0 +idna==3.18 +charset-normalizer==3.4.9 +certifi==2026.7.22 +requests==2.34.2 +seaborn==0.13.2 +uv==0.12.0 +shellingham==1.5.4 +mpmath==1.3.0 +attrs==26.1.0 +hf-xet==1.5.2 +nvidia-nccl-cu12==2.21.5 +MarkupSafe==3.0.3 +regex==2026.7.19 +importlib_metadata==9.0.0 +httpcore==1.0.9 +annotated-doc==0.0.5 +multidict==6.7.1 +aiohttp==3.14.3 +aiosignal==1.4.0 +xxhash==3.8.1 +aiohappyeyeballs==2.7.1 +mdurl==0.1.2 +cuda-toolkit==13.0.3.0 +networkx==3.6.1 +PyYAML==6.0.3 +nvidia-cufile==1.15.1.6 +typer==0.27.0 +torchaudio==2.6.0+cu124 +rich==15.0.0 +nvidia-cufft-cu12==11.2.1.3 +h11==0.16.0 +dill==0.4.1 +cuda-pathfinder==1.6.0 +filelock==3.29.0 +nvidia-nvtx-cu12==12.4.127 +httpx==0.28.1 +anyio==4.14.2 +numpy==2.4.4 +yarl==1.24.5 +click==8.4.2 +triton==3.2.0 +frozenlist==1.8.0 +zipp==4.1.0 +propcache==0.5.2 +tokenizers==0.22.2 +markdown-it-py==4.2.0 +nvidia-cuda-runtime==13.0.96 +cuda-bindings==13.3.1 +nvidia-cuda-cupti==13.0.85 +torch==2.6.0+cu124 +multiprocess==0.70.19 +pillow==12.2.0 +transformers==5.15.0.dev0 +wandb==0.28.1 +nvidia-curand==10.4.0.35 +sympy==1.13.1 +nvidia-cusparse==12.6.3.3 +nvidia-cuda-nvrtc==13.0.88 +typing-inspection==0.4.2 +nvidia-cusolver==12.0.4.66 +nvidia-cufft==12.0.0.61 +nvidia-cudnn-cu13==9.20.0.48 +nvidia-cublas==13.1.1.3 +pyarrow==25.0.0 +evaluate==0.4.6 +diffusers==0.39.0 +pydantic==2.13.4 +annotated-types==0.8.0 +protobuf==7.35.1 +sentry-sdk==2.66.1 +einops==0.8.2 +packaging==26.2 +nvidia-nvjitlink-cu12==12.4.127 +nvidia-curand-cu12==10.3.5.147 +nvidia-cusparselt-cu12==0.6.2 +nvidia-cusparse-cu12==12.3.1.170 +nvidia-cuda-runtime-cu12==12.4.127 +torchvision==0.21.0+cu124 +nvidia-cuda-nvrtc-cu12==12.4.127 +nvidia-cuda-cupti-cu12==12.4.127 +nvidia-cusolver-cu12==11.6.1.9 +nvidia-cublas-cu12==12.4.5.8 +nvidia-cudnn-cu12==9.1.0.70 +huggingface_hub==1.26.0 +datasets==5.0.1 +safetensors==0.8.0 +accelerate==1.14.0 +pydantic_core==2.46.4 +ninja==1.13.0 +autocommand==2.2.2 +backports.tarfile==1.2.0 +importlib_metadata==8.7.1 +jaraco.text==4.0.0 +jaraco.context==6.1.0 +jaraco.functools==4.4.0 +more-itertools==10.8.0 +packaging==26.0 +platformdirs==4.4.0 +tomli==2.4.0 +wheel==0.46.3 +zipp==3.23.0 diff --git a/wandb/run-20260809_055134-ppvwto7l/files/wandb-metadata.json b/wandb/run-20260809_055134-ppvwto7l/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..0f8c252ab7cc9c6a260143f5186624e58a27b859 --- /dev/null +++ b/wandb/run-20260809_055134-ppvwto7l/files/wandb-metadata.json @@ -0,0 +1,96 @@ +{ + "os": "Linux-5.15.0-126-generic-x86_64-with-glibc2.35", + "python": "CPython 3.11.15", + "startedAt": "2026-08-09T05:51:34.110899Z", + "args": [ + "--config", + "/mnt/data/zainulabideen/zain-exp/notebooks/Activation/configs/baseline1.yaml", + "--variants", + "mlp-linear-9L", + "--push" + ], + "program": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py", + "codePath": "sweep.py", + "codePathLocal": "sweep.py", + "git": { + "remote": "https://github.com/deepnevro/Activation.git", + "commit": "34b8d2e8f9a0c5751333310e69fa0c1056381deb" + }, + "email": "deepnevro@gmail.com", + "root": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation", + "host": "deeplens-k3s-node1", + "executable": "/mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python", + "cpu_count": 112, + "cpu_count_logical": 224, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1560765693952", + "used": "708272566272" + } + }, + "memory": { + "total": "2164089937920" + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-39c684a5-fde6-83d7-1663-0859795881ae" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-132944c4-b689-2b5f-89a4-d730401677ab" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bc6c3e3c-9b90-09ca-c034-774961847c54" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea" + } + ], + "cudaVersion": "12.4", + "writerId": "9oiz0iyxjogu9rr8lhkve9eqryk4v0ll" +} \ No newline at end of file diff --git a/wandb/run-20260809_055134-ppvwto7l/files/wandb-summary.json b/wandb/run-20260809_055134-ppvwto7l/files/wandb-summary.json new file mode 100644 index 0000000000000000000000000000000000000000..49f1c97bdc4f4b441dcae913d01d8e867efca440 --- /dev/null +++ b/wandb/run-20260809_055134-ppvwto7l/files/wandb-summary.json @@ -0,0 +1 @@ +{"_runtime":9,"_wandb":{"runtime":9}} \ No newline at end of file diff --git a/wandb/run-20260809_055134-ppvwto7l/logs/debug-core.log b/wandb/run-20260809_055134-ppvwto7l/logs/debug-core.log new file mode 100644 index 0000000000000000000000000000000000000000..fc89ad5818b613f72a6774144510b38eb46b1189 --- /dev/null +++ b/wandb/run-20260809_055134-ppvwto7l/logs/debug-core.log @@ -0,0 +1,20 @@ +{"time":"2026-08-09T05:51:33.863685591Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmpsns932q3/port-4000224.txt","pid":4000224,"detached":false,"idle-timeout":600000000000,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false} +{"time":"2026-08-09T05:51:33.864202947Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":4000224} +{"time":"2026-08-09T05:51:33.86420271Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-4000224-4001478-2427168253/socket","Net":"unix"}} +{"time":"2026-08-09T05:51:34.042568412Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"} +{"time":"2026-08-09T05:51:34.118653333Z","level":"INFO","msg":"handleInformInit: received","streamId":"ppvwto7l","id":"1(@)"} +{"time":"2026-08-09T05:51:34.4012495Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"ppvwto7l","id":"1(@)"} +{"time":"2026-08-09T05:51:39.812793282Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"yucvjm7hzbh6"} +{"time":"2026-08-09T05:51:44.150194715Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"yucvjm7hzbh6"} +{"time":"2026-08-09T05:51:44.547236455Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"yucvjm7hzbh6"} +{"time":"2026-08-09T05:51:44.5482416Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"ppvwto7l","id":"1(@)"} +{"time":"2026-08-09T05:51:44.549062518Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"ppvwto7l","id":"1(@)"} +{"time":"2026-08-09T05:51:44.550900825Z","level":"INFO","msg":"handleInformTeardown: server teardown initiated","id":"1(@)"} +{"time":"2026-08-09T05:51:44.550911085Z","level":"INFO","msg":"handleInformTeardown: server shutdown complete","id":"1(@)"} +{"time":"2026-08-09T05:51:44.550915633Z","level":"INFO","msg":"server: is shutting down"} +{"time":"2026-08-09T05:51:44.550916381Z","level":"INFO","msg":"connection: closing","id":"1(@)"} +{"time":"2026-08-09T05:51:44.550929029Z","level":"INFO","msg":"processOutgoingData: finished","id":"1(@)"} +{"time":"2026-08-09T05:51:44.550961866Z","level":"INFO","msg":"connection: closed successfully","id":"1(@)"} +{"time":"2026-08-09T05:51:44.550965473Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"1(@)"} +{"time":"2026-08-09T05:51:44.551073051Z","level":"INFO","msg":"server: listener closed","addr":{"Name":"/tmp/wandb-4000224-4001478-2427168253/socket","Net":"unix"}} +{"time":"2026-08-09T05:51:44.551115443Z","level":"INFO","msg":"server: all connections closed"} diff --git a/wandb/run-20260809_055134-ppvwto7l/logs/debug-internal.log b/wandb/run-20260809_055134-ppvwto7l/logs/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..03c5b6dcc957964c8687900015de0c7217403639 --- /dev/null +++ b/wandb/run-20260809_055134-ppvwto7l/logs/debug-internal.log @@ -0,0 +1,17 @@ +{"time":"2026-08-09T05:51:34.118928528Z","level":"INFO","msg":"wandb-core"} +{"time":"2026-08-09T05:51:34.119531712Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"} +{"time":"2026-08-09T05:51:34.401063124Z","level":"INFO","msg":"stream: created new stream","id":"ppvwto7l"} +{"time":"2026-08-09T05:51:34.401149755Z","level":"INFO","msg":"handler: started"} +{"time":"2026-08-09T05:51:34.401243417Z","level":"INFO","msg":"stream: started"} +{"time":"2026-08-09T05:51:34.401264715Z","level":"INFO","msg":"writer: started","stream_id":"ppvwto7l"} +{"time":"2026-08-09T05:51:34.401273087Z","level":"INFO","msg":"sender: started"} +{"time":"2026-08-09T05:51:39.847970826Z","level":"INFO","msg":"filestream: sending request","total_files":0,"uploaded_len":1} +{"time":"2026-08-09T05:51:39.944211852Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:51:44.459083269Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"} +{"time":"2026-08-09T05:51:44.459268582Z","level":"INFO","msg":"filestream: sending request","total_files":1,"uploaded_len":3,"complete":true,"exit_code":0} +{"time":"2026-08-09T05:51:44.544669385Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:51:44.546111568Z","level":"INFO","msg":"handler: operation stats","stats":{}} +{"time":"2026-08-09T05:51:44.548260935Z","level":"INFO","msg":"stream: finishing up"} +{"time":"2026-08-09T05:51:44.548274074Z","level":"INFO","msg":"handler: closed"} +{"time":"2026-08-09T05:51:44.548332472Z","level":"INFO","msg":"sender: closed"} +{"time":"2026-08-09T05:51:44.548335957Z","level":"INFO","msg":"stream: all finished"} diff --git a/wandb/run-20260809_055134-ppvwto7l/logs/debug.log b/wandb/run-20260809_055134-ppvwto7l/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..ebdcaacaa3c862be4572ac4020bf28e57d902061 --- /dev/null +++ b/wandb/run-20260809_055134-ppvwto7l/logs/debug.log @@ -0,0 +1,27 @@ +2026-08-09 05:51:34,116 INFO MainThread:4000224 [wandb_setup.py:_flush():81] Current SDK version is 0.28.1 +2026-08-09 05:51:34,116 INFO MainThread:4000224 [wandb_setup.py:_flush():81] Configure stats pid to 4000224 +2026-08-09 05:51:34,116 INFO MainThread:4000224 [wandb_setup.py:_flush():81] Loading settings from environment variables +2026-08-09 05:51:34,116 INFO MainThread:4000224 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_055134-ppvwto7l/logs/debug.log +2026-08-09 05:51:34,116 INFO MainThread:4000224 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_055134-ppvwto7l/logs/debug-internal.log +2026-08-09 05:51:34,117 INFO MainThread:4000224 [wandb_init.py:init():772] calling init triggers +2026-08-09 05:51:34,117 INFO MainThread:4000224 [wandb_init.py:init():777] wandb.init called with sweep_config: {} +config: {'_wandb': {}} +2026-08-09 05:51:34,117 INFO MainThread:4000224 [wandb_init.py:init():820] starting backend +2026-08-09 05:51:34,117 INFO MainThread:4000224 [wandb_init.py:init():835] sending inform_init request +2026-08-09 05:51:34,401 INFO MainThread:4000224 [wandb_init.py:init():840] backend started and connected +2026-08-09 05:51:34,403 INFO MainThread:4000224 [wandb_init.py:init():910] updated telemetry +2026-08-09 05:51:34,410 INFO MainThread:4000224 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout +2026-08-09 05:51:34,638 INFO MainThread:4000224 [wandb_init.py:init():978] starting run threads in backend +2026-08-09 05:51:34,712 INFO MainThread:4000224 [wandb_run.py:_console_start():2621] atexit reg +2026-08-09 05:51:34,712 INFO MainThread:4000224 [wandb_run.py:_redirect():2471] redirect: wrap_raw +2026-08-09 05:51:34,712 INFO MainThread:4000224 [wandb_run.py:_redirect():2540] Wrapping output streams. +2026-08-09 05:51:34,712 INFO MainThread:4000224 [wandb_run.py:_redirect():2563] Redirects installed. +2026-08-09 05:51:34,715 INFO MainThread:4000224 [wandb_init.py:init():1016] run started, returning control to user process +2026-08-09 05:51:34,716 INFO MainThread:4000224 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.15.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 9, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 'linear', 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'outio/mlp-linear-9L_run', 'per_device_train_batch_size': 80, 'num_train_epochs': 1, 'max_steps': 750, 'learning_rate': 0.0005, 'lr_scheduler_type': 'constant', 'lr_scheduler_kwargs': None, 'warmup_steps': 0, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.01, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 16, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-linear-9L-2.0M-20260809-055132', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 50, 'eval_delay': 0, 'per_device_eval_batch_size': 80, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': True, 'hub_token': '', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/finale-mlp-linear-9L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1} +2026-08-09 05:51:34,718 INFO MainThread:4000224 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 2001280 - > +2026-08-09 05:51:34,718 INFO MainThread:4000224 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 2001280 None +2026-08-09 05:51:44,149 INFO MainThread:4000224 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/huggingface/ppvwto7l +2026-08-09 05:51:44,149 INFO MainThread:4000224 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0 +2026-08-09 05:51:44,149 INFO MainThread:4000224 [wandb_run.py:_restore():2570] restore +2026-08-09 05:51:44,149 INFO MainThread:4000224 [wandb_run.py:_restore():2576] restore done +2026-08-09 05:51:44,547 INFO MainThread:4000224 [wandb_run.py:_footer_sync_info():3993] logging synced files diff --git a/wandb/run-20260809_055134-ppvwto7l/run-ppvwto7l.wandb b/wandb/run-20260809_055134-ppvwto7l/run-ppvwto7l.wandb new file mode 100644 index 0000000000000000000000000000000000000000..fcc49474fb779fa6fbc2788411157a0e2cb1ac39 Binary files /dev/null and b/wandb/run-20260809_055134-ppvwto7l/run-ppvwto7l.wandb differ diff --git a/wandb/run-20260809_055154-xr7l4yvo/files/config.yaml b/wandb/run-20260809_055154-xr7l4yvo/files/config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..5a4d176305ee6c1f5be957225f1f2e6e3ed7b0b7 --- /dev/null +++ b/wandb/run-20260809_055154-xr7l4yvo/files/config.yaml @@ -0,0 +1,431 @@ +_name_or_path: + value: "" +_wandb: + value: + cli_version: 0.28.1 + e: + 78a21vuu11ek41ydatzvn60d6p6iz5jh: + args: + - --config + - /mnt/data/zainulabideen/zain-exp/notebooks/Activation/configs/baseline1.yaml + - --variants + - mlp-tanh-9L + - --push + codePath: sweep.py + codePathLocal: sweep.py + cpu_count: 112 + cpu_count_logical: 224 + cudaVersion: "12.4" + disk: + /: + total: "1560765693952" + used: "708272943104" + email: deepnevro@gmail.com + executable: /mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python + git: + commit: 34b8d2e8f9a0c5751333310e69fa0c1056381deb + remote: https://github.com/deepnevro/Activation.git + gpu: NVIDIA H100 80GB HBM3 + gpu_count: 8 + gpu_nvidia: + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-39c684a5-fde6-83d7-1663-0859795881ae + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-132944c4-b689-2b5f-89a4-d730401677ab + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bc6c3e3c-9b90-09ca-c034-774961847c54 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea + host: deeplens-k3s-node1 + memory: + total: "2164089937920" + os: Linux-5.15.0-126-generic-x86_64-with-glibc2.35 + program: /mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py + python: CPython 3.11.15 + root: /mnt/data/zainulabideen/zain-exp/notebooks/Activation + startedAt: "2026-08-09T05:51:54.554743Z" + writerId: 78a21vuu11ek41ydatzvn60d6p6iz5jh + m: + - "1": train/global_step + "6": + - 3 + "7": [] + - "2": '*' + "5": 1 + "6": + - 1 + "7": [] + python_version: 3.11.15 + t: + "1": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "2": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "3": + - 2 + - 7 + - 13 + - 19 + - 66 + "4": 3.11.15 + "5": 0.28.1 + "6": 5.15.0.dev0 + "9": + "1": transformers_trainer + "12": 0.28.1 + "13": linux-x86_64 +accelerator_config: + value: + dispatch_batches: null + even_batches: true + gradient_accumulation_kwargs: null + non_blocking: false + split_batches: false + use_seedable_sampler: true +activation: + value: tanh +adam_beta1: + value: 0.9 +adam_beta2: + value: 0.999 +adam_epsilon: + value: 1e-08 +architectures: + value: null +attention_bias: + value: false +attention_dropout: + value: 0 +auto_find_batch_size: + value: false +average_tokens_across_devices: + value: true +batch_eval_metrics: + value: false +bf16: + value: true +bf16_full_eval: + value: false +bos_token_id: + value: 1 +chunk_size_feed_forward: + value: 0 +data_seed: + value: 42 +dataloader_drop_last: + value: false +dataloader_in_order: + value: true +dataloader_multiprocessing_context: + value: null +dataloader_num_workers: + value: 0 +dataloader_persistent_workers: + value: false +dataloader_pin_memory: + value: true +dataloader_prefetch_factor: + value: null +ddp_backend: + value: null +ddp_broadcast_buffers: + value: null +ddp_bucket_cap_mb: + value: null +ddp_find_unused_parameters: + value: null +ddp_static_graph: + value: null +ddp_timeout: + value: 1800 +debug: + value: [] +deepspeed: + value: null +disable_tqdm: + value: false +do_eval: + value: true +do_predict: + value: false +do_train: + value: false +dtype: + value: null +enable_jit_checkpoint: + value: false +eos_token_id: + value: 2 +eval_accumulation_steps: + value: null +eval_delay: + value: 0 +eval_do_concat_batches: + value: true +eval_on_start: + value: false +eval_steps: + value: 50 +eval_strategy: + value: steps +eval_use_gather_object: + value: false +fp16: + value: false +fp16_full_eval: + value: false +fsdp: + value: null +fsdp_config: + value: null +full_determinism: + value: false +gradient_accumulation_steps: + value: 16 +gradient_checkpointing: + value: false +gradient_checkpointing_kwargs: + value: null +greater_is_better: + value: null +head_dim: + value: 32 +hidden_act: + value: silu +hidden_size: + value: 128 +hub_always_push: + value: false +hub_model_id: + value: w-ahmad/finale-mlp-tanh-9L +hub_private_repo: + value: null +hub_revision: + value: null +hub_strategy: + value: every_save +hub_token: + value: +id2label: + value: + "0": LABEL_0 + "1": LABEL_1 +ignore_data_skip: + value: false +include_for_metrics: + value: [] +include_num_input_tokens_seen: + value: "no" +initializer_range: + value: 0.02 +intermediate_size: + value: 256 +is_encoder_decoder: + value: false +label_names: + value: null +label_smoothing_factor: + value: 0 +label2id: + value: + LABEL_0: 0 + LABEL_1: 1 +learning_rate: + value: 0.0005 +length_column_name: + value: length +liger_kernel_config: + value: null +load_best_model_at_end: + value: false +local_rank: + value: -1 +log_level: + value: passive +log_level_replica: + value: warning +log_on_each_node: + value: true +logging_first_step: + value: false +logging_nan_inf_filter: + value: true +logging_steps: + value: 20 +logging_strategy: + value: steps +lr_scheduler_kwargs: + value: null +lr_scheduler_type: + value: constant +max_grad_norm: + value: 1 +max_position_embeddings: + value: 512 +max_steps: + value: 750 +metric_for_best_model: + value: null +mlp_bias: + value: false +mlp_type: + value: mlp +model/num_parameters: + value: 2001280 +model_type: + value: tiny_llama +neftune_noise_alpha: + value: null +num_attention_heads: + value: 4 +num_hidden_layers: + value: 9 +num_key_value_heads: + value: 4 +num_train_epochs: + value: 1 +optim: + value: adamw_torch_fused +optim_args: + value: null +optim_target_modules: + value: null +output_attentions: + value: false +output_dir: + value: outio/mlp-tanh-9L_run +output_hidden_states: + value: false +pad_token_id: + value: 0 +parallelism_config: + value: null +per_device_eval_batch_size: + value: 80 +per_device_train_batch_size: + value: 80 +prediction_loss_only: + value: false +pretraining_tp: + value: 1 +problem_type: + value: null +project: + value: huggingface +push_to_hub: + value: true +remove_unused_columns: + value: false +report_to: + value: + - wandb +restore_callback_states_from_checkpoint: + value: false +resume_from_checkpoint: + value: null +return_dict: + value: true +rms_norm_eps: + value: 1e-06 +rope_parameters: + value: + rope_theta: 10000 + rope_type: default +run_name: + value: LM-mlp-tanh-9L-2.0M-20260809-055153 +save_on_each_node: + value: false +save_only_model: + value: false +save_steps: + value: 100 +save_strategy: + value: steps +save_total_limit: + value: null +seed: + value: 42 +skip_memory_metrics: + value: true +tf32: + value: null +tie_word_embeddings: + value: true +tokenizer_name: + value: w-ahmad/tiny-stories-tokenizer +torch_compile: + value: false +torch_compile_backend: + value: null +torch_compile_mode: + value: null +torch_empty_cache_steps: + value: null +trackio_bucket_id: + value: null +trackio_space_id: + value: null +trackio_static_space_id: + value: null +train_sampling_strategy: + value: random +transformers_version: + value: 5.15.0.dev0 +use_cache: + value: false +use_cpu: + value: false +use_liger_kernel: + value: false +vocab_size: + value: 4096 +warmup_steps: + value: 0 +weight_decay: + value: 0.01 diff --git a/wandb/run-20260809_055154-xr7l4yvo/files/requirements.txt b/wandb/run-20260809_055154-xr7l4yvo/files/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..123b15ebf857859624f7f4332e92341c8ef13fdf --- /dev/null +++ b/wandb/run-20260809_055154-xr7l4yvo/files/requirements.txt @@ -0,0 +1,149 @@ +asttokens==3.0.1 +comm==0.2.3 +debugpy==1.8.21 +decorator==5.3.1 +executing==2.2.1 +nest-asyncio==1.6.0 +parso==0.8.7 +platformdirs==4.11.0 +psutil==7.2.2 +ptyprocess==0.7.0 +pure_eval==0.2.3 +Pygments==2.20.0 +pyzmq==27.1.0 +setuptools==83.0.0 +six==1.17.0 +tornado==6.5.7 +traitlets==5.15.0 +fsspec==2026.4.0 +wcwidth==0.8.2 +ipython_pygments_lexers==1.1.1 +jedi==0.20.0 +jupyter_core==5.9.1 +matplotlib-inline==0.2.2 +pexpect==4.9.0 +prompt_toolkit==3.0.53 +python-dateutil==2.9.0.post0 +stack_data==0.6.3 +wheel==0.47.0 +jupyter_client==8.9.1 +pip==26.1.2 +ipython==9.15.0 +ipykernel==7.2.0 +threadpoolctl==3.6.0 +pyparsing==3.3.2 +typing_extensions==4.15.0 +Jinja2==3.1.6 +narwhals==2.24.0 +kiwisolver==1.5.0 +joblib==1.5.3 +fonttools==4.63.0 +cycler==0.12.1 +scipy==1.17.1 +pandas==3.0.5 +contourpy==1.3.3 +scikit-learn==1.9.0 +matplotlib==3.11.1 +urllib3==2.7.0 +tqdm==4.70.0 +idna==3.18 +charset-normalizer==3.4.9 +certifi==2026.7.22 +requests==2.34.2 +seaborn==0.13.2 +uv==0.12.0 +shellingham==1.5.4 +mpmath==1.3.0 +attrs==26.1.0 +hf-xet==1.5.2 +nvidia-nccl-cu12==2.21.5 +MarkupSafe==3.0.3 +regex==2026.7.19 +importlib_metadata==9.0.0 +httpcore==1.0.9 +annotated-doc==0.0.5 +multidict==6.7.1 +aiohttp==3.14.3 +aiosignal==1.4.0 +xxhash==3.8.1 +aiohappyeyeballs==2.7.1 +mdurl==0.1.2 +cuda-toolkit==13.0.3.0 +networkx==3.6.1 +PyYAML==6.0.3 +nvidia-cufile==1.15.1.6 +typer==0.27.0 +torchaudio==2.6.0+cu124 +rich==15.0.0 +nvidia-cufft-cu12==11.2.1.3 +h11==0.16.0 +dill==0.4.1 +cuda-pathfinder==1.6.0 +filelock==3.29.0 +nvidia-nvtx-cu12==12.4.127 +httpx==0.28.1 +anyio==4.14.2 +numpy==2.4.4 +yarl==1.24.5 +click==8.4.2 +triton==3.2.0 +frozenlist==1.8.0 +zipp==4.1.0 +propcache==0.5.2 +tokenizers==0.22.2 +markdown-it-py==4.2.0 +nvidia-cuda-runtime==13.0.96 +cuda-bindings==13.3.1 +nvidia-cuda-cupti==13.0.85 +torch==2.6.0+cu124 +multiprocess==0.70.19 +pillow==12.2.0 +transformers==5.15.0.dev0 +wandb==0.28.1 +nvidia-curand==10.4.0.35 +sympy==1.13.1 +nvidia-cusparse==12.6.3.3 +nvidia-cuda-nvrtc==13.0.88 +typing-inspection==0.4.2 +nvidia-cusolver==12.0.4.66 +nvidia-cufft==12.0.0.61 +nvidia-cudnn-cu13==9.20.0.48 +nvidia-cublas==13.1.1.3 +pyarrow==25.0.0 +evaluate==0.4.6 +diffusers==0.39.0 +pydantic==2.13.4 +annotated-types==0.8.0 +protobuf==7.35.1 +sentry-sdk==2.66.1 +einops==0.8.2 +packaging==26.2 +nvidia-nvjitlink-cu12==12.4.127 +nvidia-curand-cu12==10.3.5.147 +nvidia-cusparselt-cu12==0.6.2 +nvidia-cusparse-cu12==12.3.1.170 +nvidia-cuda-runtime-cu12==12.4.127 +torchvision==0.21.0+cu124 +nvidia-cuda-nvrtc-cu12==12.4.127 +nvidia-cuda-cupti-cu12==12.4.127 +nvidia-cusolver-cu12==11.6.1.9 +nvidia-cublas-cu12==12.4.5.8 +nvidia-cudnn-cu12==9.1.0.70 +huggingface_hub==1.26.0 +datasets==5.0.1 +safetensors==0.8.0 +accelerate==1.14.0 +pydantic_core==2.46.4 +ninja==1.13.0 +autocommand==2.2.2 +backports.tarfile==1.2.0 +importlib_metadata==8.7.1 +jaraco.text==4.0.0 +jaraco.context==6.1.0 +jaraco.functools==4.4.0 +more-itertools==10.8.0 +packaging==26.0 +platformdirs==4.4.0 +tomli==2.4.0 +wheel==0.46.3 +zipp==3.23.0 diff --git a/wandb/run-20260809_055154-xr7l4yvo/files/wandb-metadata.json b/wandb/run-20260809_055154-xr7l4yvo/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..d879e50dddada67671278ce0395b1c26efec1c9c --- /dev/null +++ b/wandb/run-20260809_055154-xr7l4yvo/files/wandb-metadata.json @@ -0,0 +1,96 @@ +{ + "os": "Linux-5.15.0-126-generic-x86_64-with-glibc2.35", + "python": "CPython 3.11.15", + "startedAt": "2026-08-09T05:51:54.554743Z", + "args": [ + "--config", + "/mnt/data/zainulabideen/zain-exp/notebooks/Activation/configs/baseline1.yaml", + "--variants", + "mlp-tanh-9L", + "--push" + ], + "program": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py", + "codePath": "sweep.py", + "codePathLocal": "sweep.py", + "git": { + "remote": "https://github.com/deepnevro/Activation.git", + "commit": "34b8d2e8f9a0c5751333310e69fa0c1056381deb" + }, + "email": "deepnevro@gmail.com", + "root": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation", + "host": "deeplens-k3s-node1", + "executable": "/mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python", + "cpu_count": 112, + "cpu_count_logical": 224, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1560765693952", + "used": "708272943104" + } + }, + "memory": { + "total": "2164089937920" + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-39c684a5-fde6-83d7-1663-0859795881ae" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-132944c4-b689-2b5f-89a4-d730401677ab" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bc6c3e3c-9b90-09ca-c034-774961847c54" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea" + } + ], + "cudaVersion": "12.4", + "writerId": "78a21vuu11ek41ydatzvn60d6p6iz5jh" +} \ No newline at end of file diff --git a/wandb/run-20260809_055154-xr7l4yvo/files/wandb-summary.json b/wandb/run-20260809_055154-xr7l4yvo/files/wandb-summary.json new file mode 100644 index 0000000000000000000000000000000000000000..f4ec3806c8e5abda8d70342c64624c8cf8f81552 --- /dev/null +++ b/wandb/run-20260809_055154-xr7l4yvo/files/wandb-summary.json @@ -0,0 +1 @@ +{"_wandb":{"runtime":9},"_runtime":9} \ No newline at end of file diff --git a/wandb/run-20260809_055154-xr7l4yvo/logs/debug-core.log b/wandb/run-20260809_055154-xr7l4yvo/logs/debug-core.log new file mode 100644 index 0000000000000000000000000000000000000000..f66e2a50ec4a75ed4483bfbbbafdb0469fd08662 --- /dev/null +++ b/wandb/run-20260809_055154-xr7l4yvo/logs/debug-core.log @@ -0,0 +1,20 @@ +{"time":"2026-08-09T05:51:54.304723821Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmphnmz3lmh/port-4003770.txt","pid":4003770,"detached":false,"idle-timeout":600000000000,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false} +{"time":"2026-08-09T05:51:54.305175155Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":4003770} +{"time":"2026-08-09T05:51:54.305174806Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-4003770-4005207-1456478744/socket","Net":"unix"}} +{"time":"2026-08-09T05:51:54.482042592Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"} +{"time":"2026-08-09T05:51:54.557130119Z","level":"INFO","msg":"handleInformInit: received","streamId":"xr7l4yvo","id":"1(@)"} +{"time":"2026-08-09T05:51:54.818993787Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"xr7l4yvo","id":"1(@)"} +{"time":"2026-08-09T05:52:00.223396575Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"8vvtss3n0twh"} +{"time":"2026-08-09T05:52:04.735899112Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"8vvtss3n0twh"} +{"time":"2026-08-09T05:52:05.165131163Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"8vvtss3n0twh"} +{"time":"2026-08-09T05:52:05.166076718Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"xr7l4yvo","id":"1(@)"} +{"time":"2026-08-09T05:52:05.166573614Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"xr7l4yvo","id":"1(@)"} +{"time":"2026-08-09T05:52:05.168275247Z","level":"INFO","msg":"handleInformTeardown: server teardown initiated","id":"1(@)"} +{"time":"2026-08-09T05:52:05.168291473Z","level":"INFO","msg":"handleInformTeardown: server shutdown complete","id":"1(@)"} +{"time":"2026-08-09T05:52:05.168299429Z","level":"INFO","msg":"server: is shutting down"} +{"time":"2026-08-09T05:52:05.168310652Z","level":"INFO","msg":"processOutgoingData: finished","id":"1(@)"} +{"time":"2026-08-09T05:52:05.168297313Z","level":"INFO","msg":"connection: closing","id":"1(@)"} +{"time":"2026-08-09T05:52:05.168365845Z","level":"INFO","msg":"connection: closed successfully","id":"1(@)"} +{"time":"2026-08-09T05:52:05.16837009Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"1(@)"} +{"time":"2026-08-09T05:52:05.168457232Z","level":"INFO","msg":"server: listener closed","addr":{"Name":"/tmp/wandb-4003770-4005207-1456478744/socket","Net":"unix"}} +{"time":"2026-08-09T05:52:05.168490421Z","level":"INFO","msg":"server: all connections closed"} diff --git a/wandb/run-20260809_055154-xr7l4yvo/logs/debug-internal.log b/wandb/run-20260809_055154-xr7l4yvo/logs/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..46ea5a5c38cf0764805a15628f5b764412bca51c --- /dev/null +++ b/wandb/run-20260809_055154-xr7l4yvo/logs/debug-internal.log @@ -0,0 +1,17 @@ +{"time":"2026-08-09T05:51:54.557323835Z","level":"INFO","msg":"wandb-core"} +{"time":"2026-08-09T05:51:54.557613003Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"} +{"time":"2026-08-09T05:51:54.818810378Z","level":"INFO","msg":"stream: created new stream","id":"xr7l4yvo"} +{"time":"2026-08-09T05:51:54.818911698Z","level":"INFO","msg":"handler: started"} +{"time":"2026-08-09T05:51:54.818986607Z","level":"INFO","msg":"stream: started"} +{"time":"2026-08-09T05:51:54.81900645Z","level":"INFO","msg":"writer: started","stream_id":"xr7l4yvo"} +{"time":"2026-08-09T05:51:54.819025589Z","level":"INFO","msg":"sender: started"} +{"time":"2026-08-09T05:52:00.274471268Z","level":"INFO","msg":"filestream: sending request","total_files":0,"uploaded_len":1} +{"time":"2026-08-09T05:52:00.383855621Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:52:05.064200274Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"} +{"time":"2026-08-09T05:52:05.064412769Z","level":"INFO","msg":"filestream: sending request","total_files":1,"uploaded_len":3,"complete":true,"exit_code":0} +{"time":"2026-08-09T05:52:05.162443319Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:52:05.163842042Z","level":"INFO","msg":"handler: operation stats","stats":{}} +{"time":"2026-08-09T05:52:05.166110839Z","level":"INFO","msg":"stream: finishing up"} +{"time":"2026-08-09T05:52:05.16614337Z","level":"INFO","msg":"handler: closed"} +{"time":"2026-08-09T05:52:05.166217191Z","level":"INFO","msg":"sender: closed"} +{"time":"2026-08-09T05:52:05.166221159Z","level":"INFO","msg":"stream: all finished"} diff --git a/wandb/run-20260809_055154-xr7l4yvo/logs/debug.log b/wandb/run-20260809_055154-xr7l4yvo/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..5efc73ccaef601ff761b9ab66c30bac7c71c0a49 --- /dev/null +++ b/wandb/run-20260809_055154-xr7l4yvo/logs/debug.log @@ -0,0 +1,27 @@ +2026-08-09 05:51:54,555 INFO MainThread:4003770 [wandb_setup.py:_flush():81] Current SDK version is 0.28.1 +2026-08-09 05:51:54,555 INFO MainThread:4003770 [wandb_setup.py:_flush():81] Configure stats pid to 4003770 +2026-08-09 05:51:54,555 INFO MainThread:4003770 [wandb_setup.py:_flush():81] Loading settings from environment variables +2026-08-09 05:51:54,555 INFO MainThread:4003770 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_055154-xr7l4yvo/logs/debug.log +2026-08-09 05:51:54,555 INFO MainThread:4003770 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_055154-xr7l4yvo/logs/debug-internal.log +2026-08-09 05:51:54,556 INFO MainThread:4003770 [wandb_init.py:init():772] calling init triggers +2026-08-09 05:51:54,556 INFO MainThread:4003770 [wandb_init.py:init():777] wandb.init called with sweep_config: {} +config: {'_wandb': {}} +2026-08-09 05:51:54,556 INFO MainThread:4003770 [wandb_init.py:init():820] starting backend +2026-08-09 05:51:54,556 INFO MainThread:4003770 [wandb_init.py:init():835] sending inform_init request +2026-08-09 05:51:54,819 INFO MainThread:4003770 [wandb_init.py:init():840] backend started and connected +2026-08-09 05:51:54,820 INFO MainThread:4003770 [wandb_init.py:init():910] updated telemetry +2026-08-09 05:51:54,827 INFO MainThread:4003770 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout +2026-08-09 05:51:55,067 INFO MainThread:4003770 [wandb_init.py:init():978] starting run threads in backend +2026-08-09 05:51:55,141 INFO MainThread:4003770 [wandb_run.py:_console_start():2621] atexit reg +2026-08-09 05:51:55,141 INFO MainThread:4003770 [wandb_run.py:_redirect():2471] redirect: wrap_raw +2026-08-09 05:51:55,141 INFO MainThread:4003770 [wandb_run.py:_redirect():2540] Wrapping output streams. +2026-08-09 05:51:55,141 INFO MainThread:4003770 [wandb_run.py:_redirect():2563] Redirects installed. +2026-08-09 05:51:55,144 INFO MainThread:4003770 [wandb_init.py:init():1016] run started, returning control to user process +2026-08-09 05:51:55,144 INFO MainThread:4003770 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.15.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 9, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 'tanh', 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'outio/mlp-tanh-9L_run', 'per_device_train_batch_size': 80, 'num_train_epochs': 1, 'max_steps': 750, 'learning_rate': 0.0005, 'lr_scheduler_type': 'constant', 'lr_scheduler_kwargs': None, 'warmup_steps': 0, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.01, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 16, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-tanh-9L-2.0M-20260809-055153', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 50, 'eval_delay': 0, 'per_device_eval_batch_size': 80, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': True, 'hub_token': '', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/finale-mlp-tanh-9L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1} +2026-08-09 05:51:55,146 INFO MainThread:4003770 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 2001280 - > +2026-08-09 05:51:55,146 INFO MainThread:4003770 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 2001280 None +2026-08-09 05:52:04,734 INFO MainThread:4003770 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/huggingface/xr7l4yvo +2026-08-09 05:52:04,735 INFO MainThread:4003770 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0 +2026-08-09 05:52:04,735 INFO MainThread:4003770 [wandb_run.py:_restore():2570] restore +2026-08-09 05:52:04,735 INFO MainThread:4003770 [wandb_run.py:_restore():2576] restore done +2026-08-09 05:52:05,165 INFO MainThread:4003770 [wandb_run.py:_footer_sync_info():3993] logging synced files diff --git a/wandb/run-20260809_055154-xr7l4yvo/run-xr7l4yvo.wandb b/wandb/run-20260809_055154-xr7l4yvo/run-xr7l4yvo.wandb new file mode 100644 index 0000000000000000000000000000000000000000..506a5e5d2a74340ac561d5dadfb11e6d576fcc91 Binary files /dev/null and b/wandb/run-20260809_055154-xr7l4yvo/run-xr7l4yvo.wandb differ diff --git a/wandb/run-20260809_055344-aqwnomdl/files/config.yaml b/wandb/run-20260809_055344-aqwnomdl/files/config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..2342ad0593d600965e1f78fe8bceae693902afe6 --- /dev/null +++ b/wandb/run-20260809_055344-aqwnomdl/files/config.yaml @@ -0,0 +1,433 @@ +_name_or_path: + value: "" +_wandb: + value: + cli_version: 0.28.1 + e: + umwcseb5hnjq8rd6vliqak32c3a2b0o5: + args: + - --config + - /mnt/data/zainulabideen/zain-exp/notebooks/Activation/configs/baseline1.yaml + - --variants + - mlp-linear-9L + - --push + codePath: sweep.py + codePathLocal: sweep.py + cpu_count: 112 + cpu_count_logical: 224 + cudaVersion: "12.4" + disk: + /: + total: "1560765693952" + used: "708273999872" + email: deepnevro@gmail.com + executable: /mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python + git: + commit: 34b8d2e8f9a0c5751333310e69fa0c1056381deb + remote: https://github.com/deepnevro/Activation.git + gpu: NVIDIA H100 80GB HBM3 + gpu_count: 8 + gpu_nvidia: + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-39c684a5-fde6-83d7-1663-0859795881ae + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-132944c4-b689-2b5f-89a4-d730401677ab + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bc6c3e3c-9b90-09ca-c034-774961847c54 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea + host: deeplens-k3s-node1 + memory: + total: "2164089937920" + os: Linux-5.15.0-126-generic-x86_64-with-glibc2.35 + program: /mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py + python: CPython 3.11.15 + root: /mnt/data/zainulabideen/zain-exp/notebooks/Activation + startedAt: "2026-08-09T05:53:44.447354Z" + writerId: umwcseb5hnjq8rd6vliqak32c3a2b0o5 + m: + - "1": train/global_step + "6": + - 3 + "7": [] + - "2": '*' + "5": 1 + "6": + - 1 + "7": [] + python_version: 3.11.15 + t: + "1": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "2": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "3": + - 2 + - 7 + - 13 + - 19 + - 41 + - 62 + - 66 + "4": 3.11.15 + "5": 0.28.1 + "6": 5.15.0.dev0 + "9": + "1": transformers_trainer + "12": 0.28.1 + "13": linux-x86_64 +accelerator_config: + value: + dispatch_batches: null + even_batches: true + gradient_accumulation_kwargs: null + non_blocking: false + split_batches: false + use_seedable_sampler: true +activation: + value: linear +adam_beta1: + value: 0.9 +adam_beta2: + value: 0.999 +adam_epsilon: + value: 1e-08 +architectures: + value: null +attention_bias: + value: false +attention_dropout: + value: 0 +auto_find_batch_size: + value: false +average_tokens_across_devices: + value: true +batch_eval_metrics: + value: false +bf16: + value: true +bf16_full_eval: + value: false +bos_token_id: + value: 1 +chunk_size_feed_forward: + value: 0 +data_seed: + value: 42 +dataloader_drop_last: + value: false +dataloader_in_order: + value: true +dataloader_multiprocessing_context: + value: null +dataloader_num_workers: + value: 0 +dataloader_persistent_workers: + value: false +dataloader_pin_memory: + value: true +dataloader_prefetch_factor: + value: null +ddp_backend: + value: null +ddp_broadcast_buffers: + value: null +ddp_bucket_cap_mb: + value: null +ddp_find_unused_parameters: + value: null +ddp_static_graph: + value: null +ddp_timeout: + value: 1800 +debug: + value: [] +deepspeed: + value: null +disable_tqdm: + value: false +do_eval: + value: true +do_predict: + value: false +do_train: + value: false +dtype: + value: null +enable_jit_checkpoint: + value: false +eos_token_id: + value: 2 +eval_accumulation_steps: + value: null +eval_delay: + value: 0 +eval_do_concat_batches: + value: true +eval_on_start: + value: false +eval_steps: + value: 50 +eval_strategy: + value: steps +eval_use_gather_object: + value: false +fp16: + value: false +fp16_full_eval: + value: false +fsdp: + value: null +fsdp_config: + value: null +full_determinism: + value: false +gradient_accumulation_steps: + value: 16 +gradient_checkpointing: + value: false +gradient_checkpointing_kwargs: + value: null +greater_is_better: + value: null +head_dim: + value: 32 +hidden_act: + value: silu +hidden_size: + value: 128 +hub_always_push: + value: false +hub_model_id: + value: w-ahmad/finale-mlp-linear-9L +hub_private_repo: + value: null +hub_revision: + value: null +hub_strategy: + value: every_save +hub_token: + value: +id2label: + value: + "0": LABEL_0 + "1": LABEL_1 +ignore_data_skip: + value: false +include_for_metrics: + value: [] +include_num_input_tokens_seen: + value: "no" +initializer_range: + value: 0.02 +intermediate_size: + value: 256 +is_encoder_decoder: + value: false +label_names: + value: null +label_smoothing_factor: + value: 0 +label2id: + value: + LABEL_0: 0 + LABEL_1: 1 +learning_rate: + value: 0.0005 +length_column_name: + value: length +liger_kernel_config: + value: null +load_best_model_at_end: + value: false +local_rank: + value: -1 +log_level: + value: passive +log_level_replica: + value: warning +log_on_each_node: + value: true +logging_first_step: + value: false +logging_nan_inf_filter: + value: true +logging_steps: + value: 20 +logging_strategy: + value: steps +lr_scheduler_kwargs: + value: null +lr_scheduler_type: + value: constant +max_grad_norm: + value: 1 +max_position_embeddings: + value: 512 +max_steps: + value: 750 +metric_for_best_model: + value: null +mlp_bias: + value: false +mlp_type: + value: mlp +model/num_parameters: + value: 2001280 +model_type: + value: tiny_llama +neftune_noise_alpha: + value: null +num_attention_heads: + value: 4 +num_hidden_layers: + value: 9 +num_key_value_heads: + value: 4 +num_train_epochs: + value: 1 +optim: + value: adamw_torch_fused +optim_args: + value: null +optim_target_modules: + value: null +output_attentions: + value: false +output_dir: + value: outio/mlp-linear-9L_run +output_hidden_states: + value: false +pad_token_id: + value: 0 +parallelism_config: + value: null +per_device_eval_batch_size: + value: 80 +per_device_train_batch_size: + value: 80 +prediction_loss_only: + value: false +pretraining_tp: + value: 1 +problem_type: + value: null +project: + value: huggingface +push_to_hub: + value: true +remove_unused_columns: + value: false +report_to: + value: + - wandb +restore_callback_states_from_checkpoint: + value: false +resume_from_checkpoint: + value: null +return_dict: + value: true +rms_norm_eps: + value: 1e-06 +rope_parameters: + value: + rope_theta: 10000 + rope_type: default +run_name: + value: LM-mlp-linear-9L-2.0M-20260809-055343 +save_on_each_node: + value: false +save_only_model: + value: false +save_steps: + value: 100 +save_strategy: + value: steps +save_total_limit: + value: null +seed: + value: 42 +skip_memory_metrics: + value: true +tf32: + value: null +tie_word_embeddings: + value: true +tokenizer_name: + value: w-ahmad/tiny-stories-tokenizer +torch_compile: + value: false +torch_compile_backend: + value: null +torch_compile_mode: + value: null +torch_empty_cache_steps: + value: null +trackio_bucket_id: + value: null +trackio_space_id: + value: null +trackio_static_space_id: + value: null +train_sampling_strategy: + value: random +transformers_version: + value: 5.15.0.dev0 +use_cache: + value: false +use_cpu: + value: false +use_liger_kernel: + value: false +vocab_size: + value: 4096 +warmup_steps: + value: 0 +weight_decay: + value: 0.01 diff --git a/wandb/run-20260809_055344-aqwnomdl/files/output.log b/wandb/run-20260809_055344-aqwnomdl/files/output.log new file mode 100644 index 0000000000000000000000000000000000000000..17e049168fbfb11abd1ac94bb2929ed560c87b5b --- /dev/null +++ b/wandb/run-20260809_055344-aqwnomdl/files/output.log @@ -0,0 +1,128 @@ +[transformers] `use_return_dict` is deprecated! Use `return_dict` instead! +[INFO] Causal mask (float with -inf) applied to all attention layers. +/mnt/data/zainulabideen/zain-exp/notebooks/Activation/exp.py:487: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1831.) + "std": tensor.std().item(), + 13%|█████▎ | 100/750 [03:28<18:45, 1.73s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '120.1', 'grad_norm': '19.88', 'learning_rate': '0.0005', 'epoch': '0.02696', 'train/total_time_seconds': '23.28', 'train/time_per_step_avg': '1.164', 'train/epoch_time_elapsed': '42.85', 'train/estimated_remaining_minutes': '14.16', 'train/global/act/norm': '4.587e+04', 'train/global/act/mean': '0.0008469', 'train/global/act/std': '0.406', 'train/global/act/max_abs': '8.329', 'train/global/act/frac_near_dtype_limit': '0', 'train/global/act/frac_near_user_limit': '0', 'train/global/grad/norm': '7.223', 'train/global/grad/mean': '1.586e-06', 'train/global/grad/std': '0.001277', 'train/global/grad/max_abs': '0.1865', 'train/global/grad/frac_near_dtype_limit': '0', 'train/global/grad/frac_near_user_limit': '0', 'train/global/param/norm': '56.84', 'train/global/param/mean': '0.0012', 'train/global/param/std': '0.04018', 'train/global/param/max_abs': '1', 'train/global/param/frac_near_dtype_limit': '0', 'train/global/param/frac_near_user_limit': '0', 'train/layer__model_layers_4/param/norm': '17.94', 'train/layer__model_layers_4/param/mean': '0.001547', 'train/layer__model_layers_4/param/std': '0.04425', 'train/layer__model_layers_4/param/max_abs': '1', 'train/layer__model_layers_4/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_4/param/frac_near_user_limit': '0', 'train/layer__model_layers_7/param/norm': '17.94', 'train/layer__model_layers_7/param/mean': '0.001528', 'train/layer__model_layers_7/param/std': '0.04426', 'train/layer__model_layers_7/param/max_abs': '1', 'train/layer__model_layers_7/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_7/param/frac_near_user_limit': '0', 'train/layer__model_layers_6/param/norm': '17.93', 'train/layer__model_layers_6/param/mean': '0.001632', 'train/layer__model_layers_6/param/std': '0.04423', 'train/layer__model_layers_6/param/max_abs': '1', 'train/layer__model_layers_6/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_6/param/frac_near_user_limit': '0', 'train/layer_model_layers_6/act/norm': '1.413e+04', 'train/layer_model_layers_6/act/mean': '0.001091', 'train/layer_model_layers_6/act/std': '0.4279', 'train/layer_model_layers_6/act/max_abs': '5.125', 'train/layer_model_layers_6/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_6/act/frac_near_user_limit': '0', 'train/layer_model_layers_6/grad/norm': '1.128', 'train/layer_model_layers_6/grad/mean': '-7.801e-07', 'train/layer_model_layers_6/grad/std': '0.0006964', 'train/layer_model_layers_6/grad/max_abs': '0.00766', 'train/layer_model_layers_6/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_6/grad/frac_near_user_limit': '0', 'train/layer_model_layers_7/act/norm': '1.416e+04', 'train/layer_model_layers_7/act/mean': '0.000415', 'train/layer_model_layers_7/act/std': '0.4289', 'train/layer_model_layers_7/act/max_abs': '5.5', 'train/layer_model_layers_7/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_7/act/frac_near_user_limit': '0', 'train/layer_model_layers_7/grad/norm': '1.066', 'train/layer_model_layers_7/grad/mean': '-4.246e-07', 'train/layer_model_layers_7/grad/std': '0.0006579', 'train/layer_model_layers_7/grad/max_abs': '0.008484', 'train/layer_model_layers_7/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_7/grad/frac_near_user_limit': '0', 'train/layer__model_layers_8/param/norm': '17.92', 'train/layer__model_layers_8/param/mean': '0.001427', 'train/layer__model_layers_8/param/std': '0.0442', 'train/layer__model_layers_8/param/max_abs': '1', 'train/layer__model_layers_8/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_8/param/frac_near_user_limit': '0', 'train/layer_model_layers_4/act/norm': '1.408e+04', 'train/layer_model_layers_4/act/mean': '0.000971', 'train/layer_model_layers_4/act/std': '0.4264', 'train/layer_model_layers_4/act/max_abs': '5.25', 'train/layer_model_layers_4/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_4/act/frac_near_user_limit': '0', 'train/layer_model_layers_4/grad/norm': '1.432', 'train/layer_model_layers_4/grad/mean': '-2.028e-06', 'train/layer_model_layers_4/grad/std': '0.0008836', 'train +{'loss': '102.1', 'grad_norm': '24.62', 'learning_rate': '0.0005', 'epoch': '0.05393', 'train/total_time_seconds': '41.32', 'train/time_per_step_avg': '1.033', 'train/epoch_time_elapsed': '80.11', 'train/estimated_remaining_minutes': '12.23'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '5.666', 'eval_runtime': '9.552', 'eval_samples_per_second': '997.3', 'eval_steps_per_second': '12.56', 'epoch': '0.06741', 'train/total_time_seconds': '50.11', 'train/time_per_step_avg': '1.002', 'train/epoch_time_elapsed': '108', 'train/estimated_remaining_minutes': '11.69'} +{'loss': '90.97', 'grad_norm': '18.12', 'learning_rate': '0.0005', 'epoch': '0.08089', 'train/total_time_seconds': '58.95', 'train/time_per_step_avg': '0.9824', 'train/epoch_time_elapsed': '126.4', 'train/estimated_remaining_minutes': '11.3'} +{'loss': '83.1', 'grad_norm': '40.75', 'learning_rate': '0.0005', 'epoch': '0.1079', 'train/total_time_seconds': '77.26', 'train/time_per_step_avg': '0.9657', 'train/epoch_time_elapsed': '164', 'train/estimated_remaining_minutes': '10.78'} +{'loss': '77.51', 'grad_norm': '29.62', 'learning_rate': '0.0005', 'epoch': '0.1348', 'train/total_time_seconds': '92.4', 'train/time_per_step_avg': '0.924', 'train/epoch_time_elapsed': '198.8', 'train/estimated_remaining_minutes': '10.01'} +{'eval_loss': '4.69', 'eval_runtime': '9.111', 'eval_samples_per_second': '1046', 'eval_steps_per_second': '13.17', 'epoch': '0.1348', 'train/total_time_seconds': '92.4', 'train/time_per_step_avg': '0.924', 'train/epoch_time_elapsed': '207.9', 'train/estimated_remaining_minutes': '10.01'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|█████████████████████| 1/1 [00:00<00:00, 171.58it/s] + 27%|██████████▋ | 200/750 [06:47<16:51, 1.84s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '73.3', 'grad_norm': '19.5', 'learning_rate': '0.0005', 'epoch': '0.1618', 'train/total_time_seconds': '104.5', 'train/time_per_step_avg': '0.8122', 'train/epoch_time_elapsed': '239.3', 'train/estimated_remaining_minutes': '9.144'} +{'loss': '70.38', 'grad_norm': '25.25', 'learning_rate': '0.0005', 'epoch': '0.1887', 'train/total_time_seconds': '123', 'train/time_per_step_avg': '0.8167', 'train/epoch_time_elapsed': '276.9', 'train/estimated_remaining_minutes': '8.932'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '4.264', 'eval_runtime': '9.493', 'eval_samples_per_second': '1004', 'eval_steps_per_second': '12.64', 'epoch': '0.2022', 'train/total_time_seconds': '131.9', 'train/time_per_step_avg': '0.8182', 'train/epoch_time_elapsed': '305', 'train/estimated_remaining_minutes': '8.796'} +{'loss': '68.24', 'grad_norm': '14.62', 'learning_rate': '0.0005', 'epoch': '0.2157', 'train/total_time_seconds': '141.1', 'train/time_per_step_avg': '0.822', 'train/epoch_time_elapsed': '323.6', 'train/estimated_remaining_minutes': '8.675'} +{'loss': '66.14', 'grad_norm': '13.06', 'learning_rate': '0.0005', 'epoch': '0.2427', 'train/total_time_seconds': '159.1', 'train/time_per_step_avg': '0.8184', 'train/epoch_time_elapsed': '360.9', 'train/estimated_remaining_minutes': '8.397'} +{'loss': '63.81', 'grad_norm': '18.5', 'learning_rate': '0.0005', 'epoch': '0.2696', 'train/total_time_seconds': '176.3', 'train/time_per_step_avg': '0.8385', 'train/epoch_time_elapsed': '397.6', 'train/estimated_remaining_minutes': '8.078'} +{'eval_loss': '3.934', 'eval_runtime': '9.387', 'eval_samples_per_second': '1015', 'eval_steps_per_second': '12.78', 'epoch': '0.2696', 'train/total_time_seconds': '176.3', 'train/time_per_step_avg': '0.8385', 'train/epoch_time_elapsed': '407', 'train/estimated_remaining_minutes': '8.078'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|█████████████████████| 1/1 [00:00<00:00, 157.99it/s] + 40%|████████████████ | 300/750 [10:14<14:20, 1.91s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '62.17', 'grad_norm': '16.12', 'learning_rate': '0.0005', 'epoch': '0.2966', 'train/total_time_seconds': '194.8', 'train/time_per_step_avg': '0.9025', 'train/epoch_time_elapsed': '445', 'train/estimated_remaining_minutes': '7.82'} +{'loss': '61.11', 'grad_norm': '21.25', 'learning_rate': '0.0005', 'epoch': '0.3236', 'train/total_time_seconds': '212.7', 'train/time_per_step_avg': '0.8971', 'train/epoch_time_elapsed': '482.3', 'train/estimated_remaining_minutes': '7.533'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.748', 'eval_runtime': '9.478', 'eval_samples_per_second': '1005', 'eval_steps_per_second': '12.66', 'epoch': '0.337', 'train/total_time_seconds': '222.2', 'train/time_per_step_avg': '0.9027', 'train/epoch_time_elapsed': '511.1', 'train/estimated_remaining_minutes': '7.407'} +{'loss': '60.06', 'grad_norm': '42.25', 'learning_rate': '0.0005', 'epoch': '0.3505', 'train/total_time_seconds': '229.9', 'train/time_per_step_avg': '0.8876', 'train/epoch_time_elapsed': '528.3', 'train/estimated_remaining_minutes': '7.221'} +{'loss': '59.26', 'grad_norm': '27.5', 'learning_rate': '0.0005', 'epoch': '0.3775', 'train/total_time_seconds': '248.3', 'train/time_per_step_avg': '0.8919', 'train/epoch_time_elapsed': '565.8', 'train/estimated_remaining_minutes': '6.946'} +{'loss': '58.33', 'grad_norm': '14.38', 'learning_rate': '0.0005', 'epoch': '0.4044', 'train/total_time_seconds': '266.7', 'train/time_per_step_avg': '0.9046', 'train/epoch_time_elapsed': '603.5', 'train/estimated_remaining_minutes': '6.668'} +{'eval_loss': '3.614', 'eval_runtime': '9.665', 'eval_samples_per_second': '985.7', 'eval_steps_per_second': '12.41', 'epoch': '0.4044', 'train/total_time_seconds': '266.7', 'train/time_per_step_avg': '0.9046', 'train/epoch_time_elapsed': '613.1', 'train/estimated_remaining_minutes': '6.668'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|█████████████████████| 1/1 [00:00<00:00, 115.24it/s] + 53%|█████████████████████▎ | 400/750 [13:39<10:43, 1.84s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '57.34', 'grad_norm': '32.75', 'learning_rate': '0.0005', 'epoch': '0.4314', 'train/total_time_seconds': '284.9', 'train/time_per_step_avg': '0.901', 'train/epoch_time_elapsed': '650.8', 'train/estimated_remaining_minutes': '6.38'} +{'loss': '56.7', 'grad_norm': '12.56', 'learning_rate': '0.0005', 'epoch': '0.4584', 'train/total_time_seconds': '301.7', 'train/time_per_step_avg': '0.8901', 'train/epoch_time_elapsed': '686.6', 'train/estimated_remaining_minutes': '6.064'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.493', 'eval_runtime': '9.444', 'eval_samples_per_second': '1009', 'eval_steps_per_second': '12.71', 'epoch': '0.4719', 'train/total_time_seconds': '311.1', 'train/time_per_step_avg': '0.8887', 'train/epoch_time_elapsed': '715.1', 'train/estimated_remaining_minutes': '5.925'} +{'loss': '55.84', 'grad_norm': '21.5', 'learning_rate': '0.0005', 'epoch': '0.4853', 'train/total_time_seconds': '320.2', 'train/time_per_step_avg': '0.9028', 'train/epoch_time_elapsed': '733.9', 'train/estimated_remaining_minutes': '5.781'} +{'loss': '55.38', 'grad_norm': '14.56', 'learning_rate': '0.0005', 'epoch': '0.5123', 'train/total_time_seconds': '338.5', 'train/time_per_step_avg': '0.9017', 'train/epoch_time_elapsed': '771.6', 'train/estimated_remaining_minutes': '5.493'} +{'loss': '54.94', 'grad_norm': '14.69', 'learning_rate': '0.0005', 'epoch': '0.5393', 'train/total_time_seconds': '356.6', 'train/time_per_step_avg': '0.899', 'train/epoch_time_elapsed': '809', 'train/estimated_remaining_minutes': '5.201'} +{'eval_loss': '3.416', 'eval_runtime': '9.184', 'eval_samples_per_second': '1037', 'eval_steps_per_second': '13.07', 'epoch': '0.5393', 'train/total_time_seconds': '356.6', 'train/time_per_step_avg': '0.899', 'train/epoch_time_elapsed': '818.1', 'train/estimated_remaining_minutes': '5.201'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|█████████████████████| 1/1 [00:00<00:00, 148.43it/s] + 67%|██████████████████████████▋ | 500/750 [17:04<07:50, 1.88s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '54.31', 'grad_norm': '22', 'learning_rate': '0.0005', 'epoch': '0.5662', 'train/total_time_seconds': '374.5', 'train/time_per_step_avg': '0.8963', 'train/epoch_time_elapsed': '855.8', 'train/estimated_remaining_minutes': '4.904'} +{'loss': '53.89', 'grad_norm': '22.5', 'learning_rate': '0.0005', 'epoch': '0.5932', 'train/total_time_seconds': '392.5', 'train/time_per_step_avg': '0.9075', 'train/epoch_time_elapsed': '893.4', 'train/estimated_remaining_minutes': '4.609'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.345', 'eval_runtime': '9.687', 'eval_samples_per_second': '983.5', 'eval_steps_per_second': '12.39', 'epoch': '0.6067', 'train/total_time_seconds': '401.8', 'train/time_per_step_avg': '0.9077', 'train/epoch_time_elapsed': '922', 'train/estimated_remaining_minutes': '4.465'} +{'loss': '53.53', 'grad_norm': '21.88', 'learning_rate': '0.0005', 'epoch': '0.6202', 'train/total_time_seconds': '410.9', 'train/time_per_step_avg': '0.907', 'train/epoch_time_elapsed': '940.6', 'train/estimated_remaining_minutes': '4.317'} +{'loss': '53.14', 'grad_norm': '20.62', 'learning_rate': '0.0005', 'epoch': '0.6471', 'train/total_time_seconds': '427.7', 'train/time_per_step_avg': '0.892', 'train/epoch_time_elapsed': '976.5', 'train/estimated_remaining_minutes': '4.009'} +{'loss': '52.87', 'grad_norm': '19.75', 'learning_rate': '0.0005', 'epoch': '0.6741', 'train/total_time_seconds': '445.7', 'train/time_per_step_avg': '0.8911', 'train/epoch_time_elapsed': '1014', 'train/estimated_remaining_minutes': '3.714'} +{'eval_loss': '3.297', 'eval_runtime': '9.593', 'eval_samples_per_second': '993.1', 'eval_steps_per_second': '12.51', 'epoch': '0.6741', 'train/total_time_seconds': '445.7', 'train/time_per_step_avg': '0.8911', 'train/epoch_time_elapsed': '1023', 'train/estimated_remaining_minutes': '3.714'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 94.30it/s] + 80%|████████████████████████████████ | 600/750 [20:32<04:54, 1.96s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '52.49', 'grad_norm': '31', 'learning_rate': '0.0005', 'epoch': '0.701', 'train/total_time_seconds': '464.2', 'train/time_per_step_avg': '0.8967', 'train/epoch_time_elapsed': '1061', 'train/estimated_remaining_minutes': '3.422'} +{'loss': '52.32', 'grad_norm': '16.88', 'learning_rate': '0.0005', 'epoch': '0.728', 'train/total_time_seconds': '482.5', 'train/time_per_step_avg': '0.9007', 'train/epoch_time_elapsed': '1099', 'train/estimated_remaining_minutes': '3.128'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.251', 'eval_runtime': '9.147', 'eval_samples_per_second': '1042', 'eval_steps_per_second': '13.12', 'epoch': '0.7415', 'train/total_time_seconds': '491', 'train/time_per_step_avg': '0.8914', 'train/epoch_time_elapsed': '1127', 'train/estimated_remaining_minutes': '2.976'} +{'loss': '52.02', 'grad_norm': '18.5', 'learning_rate': '0.0005', 'epoch': '0.755', 'train/total_time_seconds': '500.5', 'train/time_per_step_avg': '0.8958', 'train/epoch_time_elapsed': '1146', 'train/estimated_remaining_minutes': '2.83'} +{'loss': '51.72', 'grad_norm': '23.88', 'learning_rate': '0.0005', 'epoch': '0.7819', 'train/total_time_seconds': '518.5', 'train/time_per_step_avg': '0.9082', 'train/epoch_time_elapsed': '1184', 'train/estimated_remaining_minutes': '2.533'} +{'loss': '51.41', 'grad_norm': '17.12', 'learning_rate': '0.0005', 'epoch': '0.8089', 'train/total_time_seconds': '537.1', 'train/time_per_step_avg': '0.9136', 'train/epoch_time_elapsed': '1222', 'train/estimated_remaining_minutes': '2.238'} +{'eval_loss': '3.212', 'eval_runtime': '9.499', 'eval_samples_per_second': '1003', 'eval_steps_per_second': '12.63', 'epoch': '0.8089', 'train/total_time_seconds': '537.1', 'train/time_per_step_avg': '0.9136', 'train/epoch_time_elapsed': '1231', 'train/estimated_remaining_minutes': '2.238'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|█████████████████████| 1/1 [00:00<00:00, 152.66it/s] + 93%|█████████████████████████████████████▎ | 700/750 [23:56<01:22, 1.66s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '51.13', 'grad_norm': '18.12', 'learning_rate': '0.0005', 'epoch': '0.8359', 'train/total_time_seconds': '554.5', 'train/time_per_step_avg': '0.9037', 'train/epoch_time_elapsed': '1269', 'train/estimated_remaining_minutes': '1.938'} +{'loss': '50.86', 'grad_norm': '18.88', 'learning_rate': '0.0005', 'epoch': '0.8628', 'train/total_time_seconds': '571.9', 'train/time_per_step_avg': '0.8936', 'train/epoch_time_elapsed': '1306', 'train/estimated_remaining_minutes': '1.638'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.178', 'eval_runtime': '9.512', 'eval_samples_per_second': '1002', 'eval_steps_per_second': '12.62', 'epoch': '0.8763', 'train/total_time_seconds': '581.3', 'train/time_per_step_avg': '0.9029', 'train/epoch_time_elapsed': '1334', 'train/estimated_remaining_minutes': '1.49'} +{'loss': '50.79', 'grad_norm': '30.12', 'learning_rate': '0.0005', 'epoch': '0.8898', 'train/total_time_seconds': '590.4', 'train/time_per_step_avg': '0.8989', 'train/epoch_time_elapsed': '1353', 'train/estimated_remaining_minutes': '1.342'} +{'loss': '50.51', 'grad_norm': '23', 'learning_rate': '0.0005', 'epoch': '0.9168', 'train/total_time_seconds': '608.7', 'train/time_per_step_avg': '0.9018', 'train/epoch_time_elapsed': '1391', 'train/estimated_remaining_minutes': '1.044'} +{'loss': '50.2', 'grad_norm': '18.25', 'learning_rate': '0.0005', 'epoch': '0.9437', 'train/total_time_seconds': '625', 'train/time_per_step_avg': '0.8787', 'train/epoch_time_elapsed': '1427', 'train/estimated_remaining_minutes': '0.744'} +{'eval_loss': '3.135', 'eval_runtime': '9.387', 'eval_samples_per_second': '1015', 'eval_steps_per_second': '12.78', 'epoch': '0.9437', 'train/total_time_seconds': '625', 'train/time_per_step_avg': '0.8787', 'train/epoch_time_elapsed': '1436', 'train/estimated_remaining_minutes': '0.744'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|█████████████████████| 1/1 [00:00<00:00, 143.37it/s] + 99%|███████████████████████████████████████▍| 740/750 [25:12<00:18, 1.88s/it]/mnt/data/zainulabideen/zain-exp/notebooks/Activation/exp.py:487: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1831.) +{'loss': '49.95', 'grad_norm': '21.75', 'learning_rate': '0.0005', 'epoch': '0.9707', 'train/total_time_seconds': '643.5', 'train/time_per_step_avg': '0.8895', 'train/epoch_time_elapsed': '1474', 'train/estimated_remaining_minutes': '0.4469'} +{'loss': '49.75', 'grad_norm': '25.12', 'learning_rate': '0.0005', 'epoch': '0.9976', 'train/total_time_seconds': '661.7', 'train/time_per_step_avg': '0.898', 'train/epoch_time_elapsed': '1511', 'train/estimated_remaining_minutes': '0.149'} + "std": tensor.std().item(), +100%|████████████████████████████████████████| 750/750 [25:45<00:00, 1.99s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.108', 'eval_runtime': '9.496', 'eval_samples_per_second': '1003', 'eval_steps_per_second': '12.64', 'epoch': '1.011', 'train/total_time_seconds': '676', 'train/time_per_step_avg': '0.9477', 'train/epoch_time_elapsed': '24.85', 'train/estimated_remaining_minutes': '0', 'train/global/act/norm': '1.502e+05', 'train/global/act/mean': '-0.4277', 'train/global/act/std': '1.258', 'train/global/act/max_abs': '11.81', 'train/global/act/frac_near_dtype_limit': '0', 'train/global/act/frac_near_user_limit': '0', 'train/global/grad/norm': '4.998', 'train/global/grad/mean': '1.716e-07', 'train/global/grad/std': '0.0008833', 'train/global/grad/max_abs': '0.05859', 'train/global/grad/frac_near_dtype_limit': '0', 'train/global/grad/frac_near_user_limit': '0', 'train/global/param/norm': '75.45', 'train/global/param/mean': '0.0009513', 'train/global/param/std': '0.05332', 'train/global/param/max_abs': '1', 'train/global/param/frac_near_dtype_limit': '0', 'train/global/param/frac_near_user_limit': '0', 'train/layer__model_layers_4/param/norm': '18.83', 'train/layer__model_layers_4/param/mean': '0.001601', 'train/layer__model_layers_4/param/std': '0.04648', 'train/layer__model_layers_4/param/max_abs': '1', 'train/layer__model_layers_4/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_4/param/frac_near_user_limit': '0', 'train/layer__model_layers_7/param/norm': '19.06', 'train/layer__model_layers_7/param/mean': '0.001529', 'train/layer__model_layers_7/param/std': '0.04701', 'train/layer__model_layers_7/param/max_abs': '1', 'train/layer__model_layers_7/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_7/param/frac_near_user_limit': '0', 'train/layer__model_layers_6/param/norm': '19.3', 'train/layer__model_layers_6/param/mean': '0.001631', 'train/layer__model_layers_6/param/std': '0.0476', 'train/layer__model_layers_6/param/max_abs': '1', 'train/layer__model_layers_6/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_6/param/frac_near_user_limit': '0', 'train/layer_model_layers_6/act/norm': '2.027e+04', 'train/layer_model_layers_6/act/mean': '-0.005417', 'train/layer_model_layers_6/act/std': '0.6139', 'train/layer_model_layers_6/act/max_abs': '6.031', 'train/layer_model_layers_6/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_6/act/frac_near_user_limit': '0', 'train/layer_model_layers_6/grad/norm': '1.122', 'train/layer_model_layers_6/grad/mean': '1.056e-06', 'train/layer_model_layers_6/grad/std': '0.0006925', 'train/layer_model_layers_6/grad/max_abs': '0.009888', 'train/layer_model_layers_6/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_6/grad/frac_near_user_limit': '0', 'train/layer_model_layers_7/act/norm': '2.048e+04', 'train/layer_model_layers_7/act/mean': '-0.004186', 'train/layer_model_layers_7/act/std': '0.6203', 'train/layer_model_layers_7/act/max_abs': '7', 'train/layer_model_layers_7/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_7/act/frac_near_user_limit': '0', 'train/layer_model_layers_7/grad/norm': '0.9808', 'train/layer_model_layers_7/grad/mean': '-6.818e-08', 'train/layer_model_layers_7/grad/std': '0.0006055', 'train/layer_model_layers_7/grad/max_abs': '0.009399', 'train/layer_model_layers_7/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_7/grad/frac_near_user_limit': '0', 'train/layer__model_layers_8/param/norm': '19.81', 'train/layer__model_layers_8/param/mean': '0.001386', 'train/layer__model_layers_8/param/std': '0.0489', 'train/layer__model_layers_8/param/max_abs': '1', 'train/layer__model_layers_8/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_8/param/frac_near_user_limit': '0', 'train/layer_model_layers_4/act/norm': '1.807e+04', 'train/layer_model_layers_4/act/mean': '0.01184', 'train/layer_model_layers_4/act/std': '0.5474', 'train/layer_model_layers_4/act/max_abs': '5.219', 'train/layer_model_layers_4/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_4/act/frac_near_user_limit': '0', 'train/layer_model_layers_4/grad/norm': '1.037', 'train/layer_model_layers_4/grad/mean': '1.076e-06', 'train/layer_m + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|█████████████████████| 1/1 [00:00<00:00, 137.47it/s] +100%|████████████████████████████████████████| 750/750 [25:45<00:00, 2.06s/it] +{'train_runtime': '1546', 'train_samples_per_second': '621.1', 'train_steps_per_second': '0.485', 'train_loss': '61.38', 'epoch': '1.011', 'train/total_time_seconds': '676', 'train/time_per_step_avg': '0.9477', 'train/epoch_time_elapsed': '24.92', 'train/estimated_remaining_minutes': '0'} +100%|████████████████████████████████████████| 120/120 [00:09<00:00, 12.77it/s] +[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|█████████████████████| 1/1 [00:00<00:00, 138.50it/s] +[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|█████████████████████| 1/1 [00:00<00:00, 159.10it/s] +Found 7 files to upload + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████░░░░░░░░░░░░ 3 / 7 + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████████████████ 7 / 7 ✓ +[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|█████████████████████| 1/1 [00:00<00:00, 156.43it/s] +Found 7 files to upload + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████████████████ 7 / 7 ✓ +No files have been modified since last commit. Skipping to prevent empty commit. diff --git a/wandb/run-20260809_055344-aqwnomdl/files/requirements.txt b/wandb/run-20260809_055344-aqwnomdl/files/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..123b15ebf857859624f7f4332e92341c8ef13fdf --- /dev/null +++ b/wandb/run-20260809_055344-aqwnomdl/files/requirements.txt @@ -0,0 +1,149 @@ +asttokens==3.0.1 +comm==0.2.3 +debugpy==1.8.21 +decorator==5.3.1 +executing==2.2.1 +nest-asyncio==1.6.0 +parso==0.8.7 +platformdirs==4.11.0 +psutil==7.2.2 +ptyprocess==0.7.0 +pure_eval==0.2.3 +Pygments==2.20.0 +pyzmq==27.1.0 +setuptools==83.0.0 +six==1.17.0 +tornado==6.5.7 +traitlets==5.15.0 +fsspec==2026.4.0 +wcwidth==0.8.2 +ipython_pygments_lexers==1.1.1 +jedi==0.20.0 +jupyter_core==5.9.1 +matplotlib-inline==0.2.2 +pexpect==4.9.0 +prompt_toolkit==3.0.53 +python-dateutil==2.9.0.post0 +stack_data==0.6.3 +wheel==0.47.0 +jupyter_client==8.9.1 +pip==26.1.2 +ipython==9.15.0 +ipykernel==7.2.0 +threadpoolctl==3.6.0 +pyparsing==3.3.2 +typing_extensions==4.15.0 +Jinja2==3.1.6 +narwhals==2.24.0 +kiwisolver==1.5.0 +joblib==1.5.3 +fonttools==4.63.0 +cycler==0.12.1 +scipy==1.17.1 +pandas==3.0.5 +contourpy==1.3.3 +scikit-learn==1.9.0 +matplotlib==3.11.1 +urllib3==2.7.0 +tqdm==4.70.0 +idna==3.18 +charset-normalizer==3.4.9 +certifi==2026.7.22 +requests==2.34.2 +seaborn==0.13.2 +uv==0.12.0 +shellingham==1.5.4 +mpmath==1.3.0 +attrs==26.1.0 +hf-xet==1.5.2 +nvidia-nccl-cu12==2.21.5 +MarkupSafe==3.0.3 +regex==2026.7.19 +importlib_metadata==9.0.0 +httpcore==1.0.9 +annotated-doc==0.0.5 +multidict==6.7.1 +aiohttp==3.14.3 +aiosignal==1.4.0 +xxhash==3.8.1 +aiohappyeyeballs==2.7.1 +mdurl==0.1.2 +cuda-toolkit==13.0.3.0 +networkx==3.6.1 +PyYAML==6.0.3 +nvidia-cufile==1.15.1.6 +typer==0.27.0 +torchaudio==2.6.0+cu124 +rich==15.0.0 +nvidia-cufft-cu12==11.2.1.3 +h11==0.16.0 +dill==0.4.1 +cuda-pathfinder==1.6.0 +filelock==3.29.0 +nvidia-nvtx-cu12==12.4.127 +httpx==0.28.1 +anyio==4.14.2 +numpy==2.4.4 +yarl==1.24.5 +click==8.4.2 +triton==3.2.0 +frozenlist==1.8.0 +zipp==4.1.0 +propcache==0.5.2 +tokenizers==0.22.2 +markdown-it-py==4.2.0 +nvidia-cuda-runtime==13.0.96 +cuda-bindings==13.3.1 +nvidia-cuda-cupti==13.0.85 +torch==2.6.0+cu124 +multiprocess==0.70.19 +pillow==12.2.0 +transformers==5.15.0.dev0 +wandb==0.28.1 +nvidia-curand==10.4.0.35 +sympy==1.13.1 +nvidia-cusparse==12.6.3.3 +nvidia-cuda-nvrtc==13.0.88 +typing-inspection==0.4.2 +nvidia-cusolver==12.0.4.66 +nvidia-cufft==12.0.0.61 +nvidia-cudnn-cu13==9.20.0.48 +nvidia-cublas==13.1.1.3 +pyarrow==25.0.0 +evaluate==0.4.6 +diffusers==0.39.0 +pydantic==2.13.4 +annotated-types==0.8.0 +protobuf==7.35.1 +sentry-sdk==2.66.1 +einops==0.8.2 +packaging==26.2 +nvidia-nvjitlink-cu12==12.4.127 +nvidia-curand-cu12==10.3.5.147 +nvidia-cusparselt-cu12==0.6.2 +nvidia-cusparse-cu12==12.3.1.170 +nvidia-cuda-runtime-cu12==12.4.127 +torchvision==0.21.0+cu124 +nvidia-cuda-nvrtc-cu12==12.4.127 +nvidia-cuda-cupti-cu12==12.4.127 +nvidia-cusolver-cu12==11.6.1.9 +nvidia-cublas-cu12==12.4.5.8 +nvidia-cudnn-cu12==9.1.0.70 +huggingface_hub==1.26.0 +datasets==5.0.1 +safetensors==0.8.0 +accelerate==1.14.0 +pydantic_core==2.46.4 +ninja==1.13.0 +autocommand==2.2.2 +backports.tarfile==1.2.0 +importlib_metadata==8.7.1 +jaraco.text==4.0.0 +jaraco.context==6.1.0 +jaraco.functools==4.4.0 +more-itertools==10.8.0 +packaging==26.0 +platformdirs==4.4.0 +tomli==2.4.0 +wheel==0.46.3 +zipp==3.23.0 diff --git a/wandb/run-20260809_055344-aqwnomdl/files/wandb-metadata.json b/wandb/run-20260809_055344-aqwnomdl/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..3caa21d2671d881f823e58ea792811399d354a95 --- /dev/null +++ b/wandb/run-20260809_055344-aqwnomdl/files/wandb-metadata.json @@ -0,0 +1,96 @@ +{ + "os": "Linux-5.15.0-126-generic-x86_64-with-glibc2.35", + "python": "CPython 3.11.15", + "startedAt": "2026-08-09T05:53:44.447354Z", + "args": [ + "--config", + "/mnt/data/zainulabideen/zain-exp/notebooks/Activation/configs/baseline1.yaml", + "--variants", + "mlp-linear-9L", + "--push" + ], + "program": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py", + "codePath": "sweep.py", + "codePathLocal": "sweep.py", + "git": { + "remote": "https://github.com/deepnevro/Activation.git", + "commit": "34b8d2e8f9a0c5751333310e69fa0c1056381deb" + }, + "email": "deepnevro@gmail.com", + "root": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation", + "host": "deeplens-k3s-node1", + "executable": "/mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python", + "cpu_count": 112, + "cpu_count_logical": 224, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1560765693952", + "used": "708273999872" + } + }, + "memory": { + "total": "2164089937920" + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-39c684a5-fde6-83d7-1663-0859795881ae" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-132944c4-b689-2b5f-89a4-d730401677ab" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bc6c3e3c-9b90-09ca-c034-774961847c54" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea" + } + ], + "cudaVersion": "12.4", + "writerId": "umwcseb5hnjq8rd6vliqak32c3a2b0o5" +} \ No newline at end of file diff --git a/wandb/run-20260809_055344-aqwnomdl/files/wandb-summary.json b/wandb/run-20260809_055344-aqwnomdl/files/wandb-summary.json new file mode 100644 index 0000000000000000000000000000000000000000..c5725b26a10d8e6e09972f30d6e7007b77310e65 --- /dev/null +++ b/wandb/run-20260809_055344-aqwnomdl/files/wandb-summary.json @@ -0,0 +1 @@ +{"train/train/tensor_param_model_embed_tokens_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/std":0.022216796875,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/norm":5.34375,"train/train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_o_proj/max_abs":0.8203125,"train/train/tensor_act_model_layers_0_self_attn_v_proj/max_abs":0.98828125,"train/train/layer_model_layers_1/grad/norm":1.748869284242822,"train/train/tensor_act_model_layers_8_self_attn_q_proj/norm":12639.429603678956,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/std":0.00014220218988788194,"train/train/tensor_act_model_layers_7_mlp_up_proj/std":0.1975713855186142,"train/train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_down_proj/norm":1050.6969802776157,"train/train/tensor_param_model_norm_weight/std":0,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/std":0.0228271484375,"eval/steps_per_second":12.602,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_5_mlp_up_proj/norm":3031.677025734979,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/max_abs":0.0015869140625,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/mean":0.00026702880859375,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/std":0.0007896023136567698,"train/train/layer__model_layers_0/param/std":0.046327244641872406,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/norm":4.5,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/norm":4.5,"train/train/tensor_param_model_layers_8_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_/mean":3.085545063018799,"train/train/epoch_time_elapsed":35.902476370334625,"train/train/layer_model_layers_4/grad/mean":1.076114903002187e-06,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/std":0.025146484375,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/norm":0.01828787353542204,"train/train/tensor_act_model_layers_2_mlp_down_proj/std":0.11462433009314828,"train/train/tensor_act_model_layers_5/frac_near_dtype_limit":0,"train/train/layer_model_layers_7/grad/max_abs":0.0093994140625,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/norm":2.84375,"train/train/tensor_act_model_layers_7/mean":0.0063190460205078125,"train/train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/mean":0.02984619140625,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/mean":-0.00012033060193061829,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/std":0.0006003041614608425,"train/train/tensor_act_model_layers_7/max_abs":1.640625,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/max_abs":0.00162506103515625,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/norm":0.017496555534344645,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/max_abs":0.006103515625,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/max_abs":0.1640625,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/mean":5.274266004562378e-05,"train/train/layer_model_layers_0/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/mean":-5.602836608886719e-05,"train/train/layer__model_layers_6/param/norm":19.296900303627133,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/mean":0.00012800097465515137,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/max_abs":0.091796875,"train/train/tensor_param_model_layers_1_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_embed_tokens_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/std":0.000324684607977906,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_8/act/max_abs":6.375,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/norm":0.26025230307022634,"train/train/tensor_param_model_layers_4_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/std":0.036376953125,"train/train/tensor_act_model_layers_5_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/norm":10520.92980011775,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/norm":0.02594651082943509,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/max_abs":6.21875,"train/train/tensor_act_model_layers_0_mlp_up_proj/max_abs":1.1328125,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/mean":2.726039383560419e-06,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/std":0.0003610962964440077,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/mean":-0.0002918243408203125,"train/train/layer__model_layers_1/param/mean":0.0015187018003181064,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/mean":0.0003662109375,"train/train/tensor_act_model_layers_2_self_attn_o_proj/max_abs":0.578125,"train/train/layer__model_layers_1/param/std":0.046850783894402705,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_7_self_attn_v_proj/norm":3279.989591989278,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/max_abs":0.01092529296875,"train/train/layer_model_layers_3/grad/mean":1.3687615914854357e-06,"train/train/tensor_param_model_layers_6_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_5_post_attention_layernorm/max_abs":4.71875,"train/train/tensor_act_model_layers_7_self_attn_k_proj/mean":-0.0906982421875,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/norm":4.78125,"train/train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/std":0.00036869786499644087,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_v_proj/std":0.35742205340675426,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/std":0.0260009765625,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_post_attention_layernorm/norm":9158.569824228702,"train/train/global/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/mean":-0.000270843505859375,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_user_limit":0,"train/train/global/param/mean":0.0009513001034806474,"train/train/tensor_act_model_layers_0_mlp_down_proj/norm":1014.9781179029341,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_5/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/max_abs":0.0849609375,"train/train/tensor_act_model_norm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/max_abs":0.09033203125,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/mean":-5.939946277067065e-07,"train/train/tensor_act_model_layers_0_mlp_up_proj/mean":0.003276824951171875,"train/train/layer_model_layers_1/grad/mean":1.9777443967527144e-06,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/max_abs":0.1533203125,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/norm":5.21875,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/max_abs":0.01153564453125,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/std":0.001011064996978713,"train/train/tensor_act_model_layers_1_self_attn_q_proj/max_abs":5.5,"train/train/tensor_act_model_embed_tokens/mean":0.0006635189056396484,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/mean":-7.963180541992188e-05,"train/train/tensor_act_model_layers_5_self_attn_v_proj/norm":2959.718677812412,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_4/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn/std":0.06326528385724953,"train/train/tensor_act_model_layers_4_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/max_abs":0.0019989013671875,"train/train/layer_model_layers_8/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/std":0,"train/train/layer__model_layers_0/param/max_abs":1,"train/train/tensor_act_model_layers_5_self_attn_k_proj/mean":0.014064788818359375,"train/train/tensor_act_model_layers_5_input_layernorm/mean":0.0275726318359375,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/std":0.0021064004423660735,"train/train/tensor_act_model_layers_6/std":0.2915051843774514,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/std":0.0002978021794515658,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/mean":6.979462341405451e-07,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_6/param/mean":0.0016313573685525545,"train/train/tensor_act_model_layers_5/mean":0.008279800415039062,"train/train/tensor_act_model_layers_2_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_o_proj/mean":-0.0035886764526367188,"train/train/tensor_act_model_layers_1_self_attn_o_proj/norm":301.4580922175665,"train/train/tensor_act_model_layers_1_mlp_up_proj/mean":-0.00506591796875,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/mean":2.6694033294916153e-07,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/mean":-1.3430617400445044e-06,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/max_abs":0.083984375,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/max_abs":0.08740234375,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/max_abs":0.001708984375,"train/train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8/norm":3714.703350497436,"train/train/tensor_act_model_layers_6_self_attn/norm":798.792740246835,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/max_abs":0.0093994140625,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/mean":4.765111953020096e-06,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/max_abs":0.00347900390625,"train/train/tensor_act_model_layers_0_self_attn/mean":-0.0005276203155517578,"train/train/tensor_act_model_rotary_emb/std":0.71875,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/mean":0.0003910064697265625,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/mean":-0.0002155303955078125,"train/train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/norm":0.43539001650984194,"train/train/tensor_act_model_layers_2_mlp/mean":0.005405426025390625,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/max_abs":4.03125,"train/train/tensor_act_model_layers_5_mlp_down_proj/mean":0.0011644363403320312,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/max_abs":0.00091552734375,"train/train/layer_model_layers_4/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/mean":-0.0008254051208496094,"train/train/tensor_act_model_layers_3/frac_near_user_limit":0,"train/train/global/act/norm":150185.3480440026,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/max_abs":0.0022430419921875,"train/train/tensor_param_model_layers_7_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_0_mlp_down_proj/max_abs":0.60546875,"train/train/tensor_act_model_layers_1_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/max_abs":0.2314453125,"train/train/tensor_act_model_layers_2_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/std":0.0203857421875,"train/train/tensor_act_model_layers_8/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn/std":0.0950324238613928,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_4_self_attn_v_proj/mean":0.0008850693702697754,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/norm":3.03125,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/mean":0.0001049041748046875,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/norm":0.06249469496146752,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/mean":1.1397423804737628e-06,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp_up_proj/norm":3136.1940636118106,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/std":0.0233154296875,"train/train/tensor_act_model_layers_8_mlp_down_proj/std":0.20031891610751495,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/max_abs":0.1455078125,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_8/act/mean":-0.0016579261192908655,"train/train/tensor_act_model_layers_2_input_layernorm/std":1.0000010241923734,"train/train/tensor_param_model_layers_5_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/mean":8.378628990612924e-07,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/norm":0.018391493845650374,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/mean":8.448958396911621e-06,"train/train/tensor_act_model_layers_6_self_attn_q_proj/max_abs":6.03125,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/norm":0.6348685313220278,"train/train/tensor_act_lm_head/max_abs":11.8125,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2/mean":0.012012481689453125,"train/train/tensor_act_model_layers_2_self_attn_v_proj/mean":-0.0029439926147460938,"train/train/tensor_act_model_layers_1_mlp_down_proj/norm":1282.0529812759387,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn/frac_near_user_limit":0,"train/train/global/grad/mean":1.7156429580753228e-07,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_k_proj/std":1.1411216973846043,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/norm":5.03125,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/mean":-1.292908564209938e-06,"train/train/layer_model_layers_6/grad/max_abs":0.0098876953125,"train/train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_2/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_post_attention_layernorm/mean":-0.01399993896484375,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/max_abs":0.015380859375,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5/std":0.2813725806040801,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/max_abs":0.01129150390625,"train/train/tensor_act_model_layers_7_self_attn_q_proj/std":1.1416046069727064,"train/train/tensor_grad_model_embed_tokens_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/mean":0.009479522705078125,"train/train/tensor_grad_model_embed_tokens_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_input_layernorm/norm":9158.78973389141,"train/train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_2/grad/frac_near_user_limit":0,"train/train/layer__model_layers_3/param/std":0.04727126222105599,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_2_mlp/std":0.11462433009314828,"train/train/layer_model_layers_8/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_8/param/max_abs":1,"train/train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_0/act/std":0.5327684547972705,"train/train/layer_model_layers_5/act/mean":0.005246024865370531,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit":0,"eval/loss":3.108103036880493,"train/train/tensor_act_model_layers_7_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/mean":-0.0026712417602539062,"train/train/layer_model_layers_7/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_up_proj/max_abs":1.2109375,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn/std":0.0446176132967603,"train/train/layer__model_layers_2/param/norm":19.07513669266226,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/norm":8242.797600267277,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/std":0.00011541363059973877,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/mean":-1.5028781490400434e-06,"train/train/global/param/max_abs":1,"train/train/layer_model_layers_6/grad/std":0.0006925265544513297,"train/train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_o_proj/mean":0.0011477470397949219,"train/train/tensor_act_model_layers_6_mlp/norm":1012.1440503813626,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/norm":4.46875,"train/train/layer_model_layers_5/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_1/act/std":0.5266370859241487,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/mean":1.8550781533122063e-07,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/norm":0.6757650208040404,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/mean":-7.230043411254883e-05,"train/train/tensor_act_model_layers_2_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/max_abs":0.66015625,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/std":0.0242919921875,"train/train/tensor_act_model_layers_5_mlp_up_proj/std":0.19134540618480564,"train/train/tensor_act_model_layers_2_self_attn/max_abs":0.578125,"train/train/tensor_act_model_layers_7_self_attn_q_proj/max_abs":7,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/std":0.0006417847780511843,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/max_abs":0.00445556640625,"train/train/tensor_act_model_norm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/mean":8.419447112828502e-07,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_3/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_up_proj/max_abs":1.0859375,"train/train/tensor_act_model_layers_6_self_attn/std":0.08719287717656642,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/norm":0.028055232988969746,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_post_attention_layernorm/mean":0.03253173828125,"train/train/tensor_act_model_layers_0_mlp/max_abs":0.60546875,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/max_abs":0.005706787109375,"train/train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/norm":0.059437744968443,"train/train/tensor_act_model_layers_2/max_abs":1.328125,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_0/param/mean":0.0015775290740633531,"train/train/tensor_act_model_layers_2_post_attention_layernorm/mean":0.0318145751953125,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/norm":0.020273390128607707,"train/train/tensor_act_model_layers_8_input_layernorm/norm":9158.865478523945,"train/train/tensor_act_model_layers_4_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_post_attention_layernorm/mean":0.0382232666015625,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1/mean":0.004604339599609375,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/max_abs":0.11328125,"train/train/tensor_act_model_layers_2_self_attn_o_proj/mean":0.0020008087158203125,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/std":0.035400390625,"train/train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_o_proj/std":0.0950324238613928,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/mean":-7.686018943786621e-05,"train/train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_q_proj/mean":0.0333709716796875,"train/train/tensor_act_model_layers_4/norm":2564.812437940014,"train/train/tensor_act_model_layers_7_mlp_down_proj/std":0.12609909852702159,"train/train/tensor_act_model_layers_4_self_attn_v_proj/norm":3085.8902545445458,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/max_abs":0.003509521484375,"train/train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_2_self_attn_k_proj/max_abs":4.75,"train/train/tensor_act_model_layers_1_self_attn_k_proj/norm":7514.913359677975,"train/train/tensor_act_model_layers_8_input_layernorm/std":1.000000241678179,"train/train/tensor_act_model_layers_3_self_attn_o_proj/max_abs":0.47265625,"train/train/tensor_act_model_layers_1_mlp/norm":1282.0529812759387,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/mean":-0.0002193450927734375,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_2/param/std":0.04707263888163651,"train/train/tensor_act_model_layers_4_self_attn_k_proj/mean":0.017276763916015625,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/mean":-9.489059448242188e-05,"train/train/layer_model_layers_0/act/norm":17591.951382313935,"train/train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/std":0.0013198556532964516,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/mean":1.2099742889404297e-05,"train/train/tensor_act_model_layers_4_mlp_up_proj/mean":-0.000986814498901367,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp/mean":9.156763553619385e-06,"train/train/tensor_act_model_layers_5_post_attention_layernorm/norm":9158.866821301317,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_lm_head/mean":-2.0576171875000004,"train/train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_norm_weight/std":0.002068276652680219,"train/train/tensor_act_model_layers_3/max_abs":1.484375,"train/train/tensor_param_model_norm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/norm":9158.862487805487,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/global_step":750,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/mean":-0.0002651214599609375,"train/train/tensor_param_model_layers_8_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_1_mlp/std":0.13964852814435932,"train/train/tensor_act_/std":0,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_7/act/mean":-0.004186410170335036,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/std":0.0255126953125,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/max_abs":0.027099609375,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/max_abs":0.091796875,"train/train/tensor_act_model_layers_2_self_attn_v_proj/norm":2736.5449248623486,"train/train/layer_model_layers_8/grad/mean":1.9631800408854315e-07,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/max_abs":0.1533203125,"train/train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/norm":0.38665123552465125,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/max_abs":0.004791259765625,"train/train/tensor_act_model_layers_2_self_attn_k_proj/mean":0.015628814697265625,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/norm":4.84375,"train/train/tensor_param_model_norm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_input_layernorm/norm":9158.515502939657,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/std":0.0007855715454180655,"train/train/tensor_act_model_layers_3/mean":0.00946044921875,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/max_abs":0.00142669677734375,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/std":0.035400390625,"train/train/tensor_act_model_layers_5_post_attention_layernorm/mean":0.0240020751953125,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/std":0.0013908427641118966,"train/train/tensor_act_model_layers_8_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/mean":0.02191925048828125,"train/train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/std":0.0007164730821052101,"train/train/layer_model_layers_7/grad/norm":0.9808154580741602,"train/train/tensor_act_model_layers_8_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/max_abs":0.1494140625,"train/train/tensor_act_model_layers_8_self_attn/norm":1078.046025070771,"train/train/tensor_act_model_layers_3_mlp_up_proj/norm":3097.754188140022,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/mean":-4.673004150390625e-05,"train/train/layer_model_layers_5/grad/max_abs":0.01092529296875,"train/train/layer_model_layers_7/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_0/grad/max_abs":0.0286865234375,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/norm":2.921875,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp/max_abs":0.55859375,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/std":0.023681640625,"train/train/tensor_act_model_layers_4_input_layernorm/std":1.0000005615582506,"train/train/tensor_act_model_norm/std":1.0000002746237064,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/max_abs":0.080078125,"train/train/tensor_grad_model_embed_tokens_weight/norm":1.3963682753312472,"train/train/layer_model_layers_0/grad/norm":3.358869449977567,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/max_abs":0.1748046875,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/mean":-0.00017432868480682373,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/max_abs":0.00927734375,"train/train/tensor_act_model_layers_7_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/norm":4.5,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/norm":0.16623753694556406,"train/train/tensor_act_model_layers_5_self_attn_v_proj/max_abs":1.84375,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/std":0.037353515625,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/norm":0.30735278195334903,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/norm":0.02808577331898204,"train/train/tensor_param_model_layers_3_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_v_proj/mean":0.01746368408203125,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/max_abs":0.00823974609375,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/mean":-0.00016880035400390625,"train/train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_down_proj/max_abs":0.6171875,"train/train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0/std":0.11718775424792081,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/norm":0.32410548426276165,"train/train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/max_abs":0.007232666015625,"train/train/tensor_act_model_layers_0_post_attention_layernorm/std":1.0000004657698636,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/norm":0.018823162034157223,"train/train/tensor_act_model_layers_7/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/max_abs":0.001190185546875,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/std":0.00012209175806172248,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/max_abs":0.1201171875,"train_steps_per_second":0.485,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/norm":0.07685140577189588,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/std":0.00047807733058092964,"train/train/tensor_act_model_layers_5_post_attention_layernorm/std":1.00000024016478,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/norm":0.7119194208474616,"train/train/tensor_act_/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/std":0.0008503758163232629,"train/train/tensor_act_model_layers_6_self_attn/max_abs":1.046875,"train/train/tensor_act_model_layers_6_post_attention_layernorm/norm":9158.868591321168,"train/train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/norm":0.019447025851964624,"train/train/layer_model_layers_5/grad/frac_near_user_limit":0,"train/train/layer_model_layers_2/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_up_proj/mean":2.9494520276784897e-05,"train/train/tensor_act_model_layers_8_post_attention_layernorm/norm":9158.873291019005,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_post_attention_layernorm/norm":9158.794067390949,"train/train/tensor_param_model_embed_tokens_weight/std":0.06591796875,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/norm":5.65625,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_up_proj/std":0.19531258106913943,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/mean":9.60018951445818e-07,"train/train/tensor_act_model_layers_4_post_attention_layernorm/norm":9158.856872565368,"train/train/layer_model_layers_2/act/norm":18632.33940555953,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_1/param/norm":18.983731984995075,"train/train/tensor_act_model_layers_8_mlp_up_proj/max_abs":1.3671875,"train/train/tensor_act_model_layers_8_mlp/max_abs":1.1640625,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/mean":-0.0001233704388141632,"train/train/tensor_act_model_layers_8_self_attn_v_proj/max_abs":1.921875,"train/train/tensor_act_model_layers_8_self_attn/mean":0.0038623809814453125,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/std":0.00016652373622645598,"train/train/tensor_act_model_layers_4_mlp_up_proj/frac_near_user_limit":0,"train/train/layer_model_layers_7/act/std":0.6202687217424423,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/mean":6.784126162528992e-05,"train/train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_user_limit":0,"train_samples_per_second":621.063,"train/train/tensor_act_model_layers_8_self_attn_v_proj/std":0.37829724816102006,"train/train/tensor_act_model_layers_4_self_attn_o_proj/norm":580.2747154684458,"train/train/tensor_act_model_layers_5_self_attn_k_proj/norm":8086.794526499998,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/mean":-0.0014693737030029297,"train/train/tensor_act_model_layers_5_self_attn_v_proj/mean":-0.003086090087890625,"train/train/layer_model_layers_5/grad/norm":1.1070405366170881,"train/train/tensor_act_model/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/max_abs":0.0038604736328125,"train/train/tensor_act_model_layers_3_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/norm":0.6969099232771115,"train/train/tensor_act_model_layers_5/max_abs":1.5,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/max_abs":0.00140380859375,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/mean":-1.801163307391107e-06,"train/train/tensor_act_model_layers_6_self_attn_o_proj/max_abs":1.046875,"train/train/tensor_act_model_layers_6_mlp_down_proj/norm":1012.1440503813626,"train/train/tensor_act_model_layers_2_self_attn_o_proj/norm":408.9549750343595,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/max_abs":0.004425048828125,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/norm":0.330183565352083,"train/train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_1/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/mean":1.3037933968007565e-06,"train/train/tensor_act_model_layers_7_self_attn_k_proj/norm":10485.176929176645,"train/train/tensor_act_model_rotary_emb/frac_near_dtype_limit":0,"total_flos":4.354347480907776e+15,"train/train/tensor_act_model_layers_7_self_attn_q_proj/mean":-0.0406341552734375,"train/train/layer_model_layers_6/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_6/act/mean":-0.005417345808102534,"train/train/layer_model_layers_3/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/max_abs":0.007598876953125,"train/train/estimated_remaining_minutes":0,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/mean":-6.993068382143974e-07,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/norm":3.015625,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/max_abs":0.1494140625,"train/train/layer_model_layers_4/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_8/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_input_layernorm/max_abs":4.46875,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/norm":0.09609577828104279,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/max_abs":0.00147247314453125,"train/train/tensor_act_model_layers_2_mlp_up_proj/frac_near_user_limit":0,"train/train/global/act/std":1.2584756013799252,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_q_proj/norm":10739.079023412945,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_0/param/norm":18.78089251947641,"train/train/tensor_act_model_layers_0_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/max_abs":0.125,"train/train/tensor_param_model_layers_2_input_layernorm_weight/mean":1,"train/train/layer_model_layers_6/grad/norm":1.1219461188048783,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/norm":1.4804172154618904,"train/epoch":1.0107853050219076,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/norm":4.65625,"train/train/layer_model_layers_7/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/std":0.10058596968487546,"train/train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/std":0.0274658203125,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/max_abs":0.00689697265625,"train/train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/norm":9158.321228096187,"train/train/tensor_act_model_layers_2_input_layernorm/mean":0.02327728271484375,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/mean":-1.1431053280830383e-05,"train/train/tensor_act_model_layers_2_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_3_mlp/max_abs":0.66015625,"train/train/tensor_param_model_layers_5_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_3_post_attention_layernorm/norm":9158.835632337961,"train/train/layer_model_layers_2/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/mean":-2.082379069179297e-06,"train/train/tensor_act_model_layers_4_mlp/norm":921.3529462980846,"train/train/layer__model_layers_3/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/max_abs":4.875,"train/train/tensor_act_model_layers_7_self_attn_k_proj/max_abs":5.21875,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_3_post_attention_layernorm/max_abs":4.8125,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/mean":4.839897155761719e-05,"train/train/tensor_act_model_layers_4_self_attn/mean":0.0011477470397949219,"train/train/tensor_act_model_layers_6_mlp_up_proj/norm":3078.6657922183363,"train/train/tensor_act_model_layers_3/norm":2475.276032622091,"train/train/layer_model_layers_4/act/norm":18073.148360412022,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/max_abs":0.00537109375,"train/train/tensor_act_model_layers_7_mlp_down_proj/mean":-0.0019550323486328125,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_8_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn/std":0.11758523080684756,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/std":0.0004835319855374909,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/norm":4.75,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_o_proj/max_abs":0.35546875,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/global/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/max_abs":0.002655029296875,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/max_abs":0.08251953125,"train/train/global/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_post_attention_layernorm/mean":0.04740905761718751,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/norm":0.01664093901557326,"train/train/tensor_act_model_layers_2_self_attn_q_proj/max_abs":5.25,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/std":0.00020012373680981202,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/mean":1.9668368622660633e-06,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/std":0.0010473705549273293,"train/train/tensor_act_model_layers_6/mean":0.011852264404296877,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/norm":0.0906099734509409,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/mean":-1.7615966498851776e-06,"train/train/tensor_act_model_layers_2_mlp/max_abs":0.66796875,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/max_abs":0.00592041015625,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/norm":0.03215523252833009,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/std":0.04248046875,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/max_abs":0.0859375,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/std":0.0308837890625,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_up_proj/std":0.19018606370318258,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/max_abs":0.125,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_5/grad/mean":3.8479168589885073e-07,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/norm":5.28125,"train/train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_3/param/max_abs":1,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/mean":4.579778760671616e-06,"train/train/tensor_act_model_layers_8_mlp_down_proj/norm":1833.4078311255546,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp_down_proj/std":0.13964852814435932,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp/std":0.12609909852702159,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/std":0.00041555377899105605,"train/train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_q_proj/std":1.3794029002653125,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/norm":4.46875,"train/train/global/grad/std":0.0008832509986677723,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_1_self_attn_v_proj/mean":0.0036802291870117188,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/std":0.000634546290983288,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/std":0.00022058114423191606,"train/train/layer_model_layers_0/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/max_abs":0.166015625,"train/train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/norm":4.5,"train/train/tensor_act_model_layers_8_self_attn_k_proj/std":1.293951040201286,"train/train/layer__model_layers_7/param/mean":0.001528687856498635,"train/train/tensor_act_model_layers_6_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/std":0.00012644644580173545,"train/train/layer_model_layers_2/grad/std":0.0008341103202784663,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/norm":4.59375,"train/train/tensor_param_model_layers_0_input_layernorm_weight/mean":1,"train/train/layer_model_layers_2/act/std":0.5642383070261626,"train/train/tensor_act_model_layers_5_mlp/mean":0.0011644363403320312,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/mean":7.05718994140625e-05,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/max_abs":0.004608154296875,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/mean":-0.00028228759765625,"train/train/layer_model_layers_7/grad/mean":-6.818339608568995e-08,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp_up_proj/norm":3753.7269197260766,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/norm":1.7430773459698234,"train/train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/mean":6.8247318267822266e-06,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_5_input_layernorm/norm":9158.861267098808,"train/train/tensor_act_model_layers_3_mlp_down_proj/std":0.10897847648259888,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/mean":3.027264028787613e-06,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/max_abs":0.1806640625,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/max_abs":0.5390625,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_2/grad/norm":1.3519265251055885,"train/train/tensor_act_model_layers_4_mlp/mean":-0.0026712417602539062,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_4/param/mean":0.0016009677404918462,"train/train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/mean":-0.0022554397583007812,"train/train/layer_model_layers_6/act/norm":20266.952184443417,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/max_abs":0.0040283203125,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/max_abs":0.0028076171875,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/mean":4.1157007217407227e-05,"train/train/tensor_act_model_layers_1_self_attn_k_proj/mean":0.047149658203125,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/norm":4.5625,"train/train/tensor_act_model_norm/mean":0.0232391357421875,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/std":0.02001953125,"train/train/layer_model_layers_7/act/norm":20479.295498580417,"train/train/tensor_act_model_layers_1_self_attn_v_proj/norm":2493.6288743965956,"train/train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/max_abs":0.005889892578125,"train/train/layer_model_layers_2/grad/mean":2.437561762035358e-06,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_8/param/mean":0.0013863359710169657,"train/train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_o_proj/std":0.11758523080684756,"train/train/tensor_act_model_layers_6_self_attn_k_proj/max_abs":4.40625,"train/train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/mean":3.9157457649707794e-05,"train/train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/max_abs":0.08642578125,"train/train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_4/act/frac_near_user_limit":0,"train/train/tensor_act_/norm":12.34277911184733,"train/train/tensor_act_model_layers_4_self_attn_v_proj/max_abs":1.7734375,"train/train/tensor_act_model_layers_5_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_embed_tokens/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp/max_abs":0.6171875,"train/train/tensor_act_model_layers_7/std":0.29797527457897055,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/max_abs":0.08154296875,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/max_abs":0.20703125,"train/train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_input_layernorm/mean":0.003910541534423829,"train/train/global/act/mean":-0.42774814351682083,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/norm":1.1280915907881908,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/norm":0.044414914575084743,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp/max_abs":0.88671875,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/norm":4.6875,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/mean":-7.91158527135849e-06,"train/train/layer__model_layers_7/param/max_abs":1,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/mean":0.00012053549289703369,"train/train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/std":0.00029343398646999646,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/std":0.0034055304413431733,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/norm":0.014499166008893285,"train/train/tensor_act_model_layers_8/max_abs":2.03125,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_7/param/std":0.04701331013729768,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/norm":0.32488650761207233,"train/train/tensor_act_model_layers_2_self_attn_q_proj/mean":0.016788482666015625,"train/train/layer__model_layers_6/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_k_proj/norm":7735.051202050646,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_o_proj/norm":870.6736716013772,"train/train/tensor_act_model_layers_1_self_attn_o_proj/mean":-0.0022516250610351562,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_6_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_embed_tokens/std":0.07476825003052505,"train/train/global/param/std":0.05331836449904397,"train/train/tensor_param_model_layers_7_input_layernorm_weight/std":0,"train/train/tensor_param_model_norm_weight/mean":1,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/norm":2.984375,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/norm":5.4375,"train/train/tensor_act_model_layers_3_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/norm":0.7933408869322305,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/std":0.0009209493170190383,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/mean":1.6889534890651703e-06,"train/train/tensor_act_model_layers_7_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_q_proj/mean":0.062774658203125,"train/train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_5/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_4_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/std":0.0203857421875,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/mean":-0.000213623046875,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2/norm":2229.1012974584337,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/std":0.0005738216679627089,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/max_abs":0.2265625,"train/train/tensor_act_model_layers_4_self_attn_k_proj/norm":7745.710212624892,"train/train/layer_model_layers_8/act/std":0.6887773225057738,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/mean":8.106231689453125e-05,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/std":0.0006846107163238569,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/mean":3.46451997756958e-06,"train/train/layer__model_layers_8/param/std":0.04890227832752276,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/norm":4.875,"train/train/tensor_act_model_layers_3_self_attn/max_abs":0.47265625,"train/train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/mean":2.1673622541129593e-06,"train/train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_input_layernorm/mean":0.021881103515625003,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/norm":0.11293648181112204,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/mean":8.337037797900848e-07,"train/train/tensor_act_model_layers_1_self_attn/max_abs":0.35546875,"train/train/layer_model_layers_1/grad/max_abs":0.01153564453125,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/max_abs":0.0034332275390625,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/norm":3.953125,"train/train/tensor_act_model_layers_5_self_attn_o_proj/norm":497.6350976881482,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/norm":0.3505199657200831,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_7/act/max_abs":7,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_post_attention_layernorm/max_abs":4.59375,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/std":0.020751953125,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/max_abs":0.007171630859375,"train/train/tensor_act_model_layers_4_self_attn/frac_near_dtype_limit":0,"train/train/layer__model_layers_0/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/mean":1.3246608432382343e-06,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/norm":1.0471109152906861,"train/train/tensor_act_lm_head/frac_near_user_limit":0,"train/train/global/grad/max_abs":0.05859375,"train/train/tensor_act_model_rotary_emb/norm":3632.2067871093745,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/norm":0.47152372510489016,"train/train/tensor_act_model_layers_1_self_attn_k_proj/max_abs":4.84375,"train/train/tensor_act_model_layers_0_mlp_down_proj/std":0.11087058315602684,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/std":0.0208740234375,"train/train/tensor_act_model_layers_2_post_attention_layernorm/std":1.0000009535574466,"train/train/layer__model_layers_3/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model/max_abs":4.75,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/max_abs":0.00244140625,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/norm":3.328125,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/mean":-1.8938444554805756e-06,"train/train/tensor_act_model_layers_5_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/mean":-1.1397060006856918e-06,"train/train/layer__model_layers_8/param/norm":19.81155114052784,"train/train/tensor_act_model_layers_6_self_attn_q_proj/std":1.172365214812159,"train/train/tensor_act_model_layers_6_self_attn_o_proj/mean":-0.002144336700439453,"train/train/tensor_act_model_layers_0_self_attn_k_proj/std":0.8269067291263272,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/mean":0.00057220458984375,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/max_abs":0.224609375,"train/train/tensor_act_model_layers_5_self_attn_q_proj/max_abs":6,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/mean":-0.0001430511474609375,"train/train/tensor_act_model_layers_4_mlp_up_proj/frac_near_dtype_limit":0,"train/train/global/act/max_abs":11.8125,"train/train/tensor_act_model_layers_7_post_attention_layernorm/max_abs":4.6875,"train/train/tensor_act_model_layers_8_self_attn_v_proj/mean":-0.0057849884033203125,"train/train/layer__model_layers_7/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/max_abs":0.0020599365234375,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/max_abs":0.005035400390625,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit":0,"_timestamp":1.786256380850631e+09,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/std":0.0002273880887448863,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/max_abs":0.08642578125,"train/train/tensor_act_model_layers_3_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_0/act/mean":-0.0008019591824939618,"train/train/tensor_act_model_layers_2_self_attn_v_proj/std":0.29882832015005995,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/norm":0.2207897995078383,"train/train/tensor_act_model_layers_0_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0/max_abs":0.72265625,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/max_abs":0.00628662109375,"train/train/layer_model_layers_6/act/std":0.6138518434558304,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/std":0.8610920463597765,"train/train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/mean":-1.9311904907226562e-05,"train/train/tensor_act_model_layers_2_mlp_up_proj/norm":3125.8167549285604,"train/train/layer_model_layers_0/act/max_abs":6.21875,"train/train/tensor_act_model_layers_4_mlp_down_proj/norm":921.3529462980846,"train/train/tensor_act_model_layers_5_input_layernorm/std":1.0000002960441194,"train/train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/std":0.0211181640625,"train/train/layer_model_layers_8/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn/mean":-0.0035886764526367188,"train/train/layer_model_layers_8/grad/max_abs":0.009765625,"train/train/tensor_act_model_layers_3_self_attn_q_proj/std":1.138189614442459,"train/train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_k_proj/mean":-0.085968017578125,"train/train/layer__model_layers_5/param/mean":0.001528687856498635,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/norm":0.5816282374777465,"train/train/tensor_act_model_layers_6_self_attn_v_proj/norm":3351.5042061824015,"train/train/tensor_act_model_layers_6_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_mlp/std":0.1015626140108337,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_input_layernorm/norm":9158.859802255498,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/std":0.0011359983194955572,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/std":0.03564453125,"train/train/tensor_act_model_layers_8_mlp_up_proj/std":0.23651187933555926,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_up_proj/max_abs":1.1171875,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/std":0.037841796875,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/max_abs":0.00494384765625,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/std":0.0007156685057058643,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_7_input_layernorm/mean":0.04217529296875,"train/train/tensor_act_model_layers_0/mean":0.0001455782912671566,"train/train/tensor_act_model_layers_6_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_4/grad/std":0.0006396265473420889,"train/train/tensor_act_model_layers_3_mlp_down_proj/norm":998.9542597675202,"train/train/tensor_act_model_layers_0/frac_near_user_limit":0,"train/train/layer_model_layers_2/grad/max_abs":0.007598876953125,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_k_proj/max_abs":5.15625,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/mean":7.486343383789062e-05,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/norm":0.016322047857146693,"train/train/tensor_act_model_layers_2_self_attn_v_proj/max_abs":1.8671875,"train/train/tensor_grad_model_norm_weight/norm":0.3140373561530066,"train/train/tensor_act_model_layers_5_mlp_down_proj/max_abs":0.55859375,"train/train/tensor_act_model/std":1.0000002746237064,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/max_abs":0.0019989013671875,"train/train/layer_model_layers_3/act/mean":0.0260999844624446,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/norm":3.046875,"train/train/tensor_act_model_layers_4_self_attn/norm":580.2747154684458,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp_down_proj/max_abs":0.88671875,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/norm":0.6962897473261009,"train/train/time_per_step_avg":0.9476513889431953,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_k_proj/norm":11854.747864460993,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit":0,"eval/samples_per_second":1000.475,"train/train/tensor_act_model_layers_8_self_attn_q_proj/mean":-0.048736572265625,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model/norm":9158.885131843323,"train/train/tensor_param_model_layers_3_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/max_abs":0.002410888671875,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_0_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_input_layernorm/mean":0.04978942871093751,"train/train/tensor_act_model_layers_8/std":0.40576329341056694,"train/train/tensor_act_model_layers_4/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_post_attention_layernorm/max_abs":4.96875,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/norm":0.03569999716599103,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/std":0.037841796875,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/std":0.0006330197896350134,"train/train/tensor_act_model_layers_5_self_attn/max_abs":0.4375,"train/train/tensor_param_model_layers_5_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_8_post_attention_layernorm/max_abs":4.71875,"train/train/tensor_act_model_layers_3_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/mean":-0.00018024444580078125,"train/train/tensor_act_model_layers_7_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/mean":0.0001621246337890625,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/std":0.0005453423260938054,"train/train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/max_abs":0.006072998046875,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn/mean":-0.002144336700439453,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_8/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_down_proj/max_abs":0.66796875,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/norm":0.02216423019948128,"train/train/layer_model_layers_1/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_5/act/max_abs":6,"train/train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_input_layernorm/max_abs":4.53125,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/std":0.0009775568531857459,"train/train/tensor_act_model_layers_4_self_attn/max_abs":0.60546875,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/std":0.020751953125,"train/train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/norm":4.40625,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/max_abs":0.10888671875,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/mean":1.3120006769895554e-07,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/norm":0.520757618804886,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/max_abs":0.00193023681640625,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/std":0.0005979014626018539,"train/train/layer_model_layers_3/act/std":0.5813081153399244,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/std":0.00036285050479602693,"train/train/tensor_act_model_layers_3_self_attn_q_proj/mean":0.15374755859375,"train/train/tensor_act_model_layers_3_self_attn_k_proj/norm":7931.644892515877,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/max_abs":0.00714111328125,"train/train/tensor_act_model_layers_3_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/std":0.0004466476153536851,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/mean":1.2395321391522884e-07,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/mean":5.408190190792084e-05,"train_loss":61.381723103841146,"train/train/layer__model_layers_8/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/norm":4.625,"train/train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/norm":0.7220449800719276,"train/train/tensor_act_model_layers_0_self_attn_v_proj/mean":-0.0010931491851806638,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/max_abs":0.002716064453125,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/norm":3.265625,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/std":0.0011805190822957745,"train/train/tensor_act_model_layers_8_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/std":0.0006448974090784716,"train/train/layer_model_layers_0/grad/std":0.002073075775157821,"train/train/tensor_act_model_layers_0_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/mean":-6.246566772460938e-05,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_post_attention_layernorm/std":1.0000002372252939,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/std":0.0247802734375,"train/train/layer_model_layers_0/act/frac_near_dtype_limit":0,"train/train/layer__model_layers_1/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_1_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/norm":0.08525953765832373,"train/train/layer_model_layers_5/act/norm":18287.03101790653,"train/train/tensor_act_model_layers_5/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/std":0.040771484375,"train/train/tensor_act_model_layers_6_mlp_up_proj/mean":0.0003501623868942261,"train/train/layer_model_layers_3/act/norm":19206.574143126698,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_up_proj/std":0.19427530100764198,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_3/param/mean":0.0015184280466921801,"train/train/tensor_act_model_layers_6_post_attention_layernorm/std":1.000000366911923,"train/train/tensor_act_model_layers_4_mlp/max_abs":0.5390625,"train/train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/mean":0.00026702880859375,"train/train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_2/param/max_abs":1,"train/train/tensor_act_model_layers_6_self_attn_o_proj/norm":798.792740246835,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/norm":3.109375,"train/train/tensor_act_model_layers_2_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/norm":4.84375,"train/train/layer_model_layers_1/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/mean":-0.000148773193359375,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/max_abs":0.1142578125,"train/train/tensor_act_model_layers_6_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn/std":0.01518657306789068,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/norm":2.921875,"train/train/tensor_act_model_layers_5_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_q_proj/max_abs":6.09375,"train/train/tensor_act_model_layers_1_input_layernorm/max_abs":4.875,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_user_limit":0,"train_runtime":1545.7365,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/std":0.0008954833779846271,"train/train/tensor_act_model_layers_4_self_attn_o_proj/std":0.06326528385724953,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/mean":-2.3588654585182667e-07,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/max_abs":0.0020751953125,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/std":0.0005966479081292068,"train/train/tensor_act_model_layers_4_mlp_up_proj/max_abs":1.1015625,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/std":0.020263671875,"train/train/layer_model_layers_2/act/max_abs":5.25,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/max_abs":0.1044921875,"train/train/tensor_act_model_layers_2_self_attn_q_proj/norm":9736.292863798752,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/mean":-1.889653503894806e-06,"train/train/tensor_act_model_layers_5_self_attn/std":0.05436818972384846,"train/train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/mean":-2.587307244539261e-05,"train/train/tensor_act_lm_head/frac_near_dtype_limit":0,"train/train/layer__model_layers_4/param/std":0.04647544995312572,"train/train/tensor_grad_model_norm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp/norm":1014.9781179029341,"train/train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/norm":0.11642218871094213,"train/train/tensor_grad_model_norm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_user_limit":0,"_wandb":{"runtime":1557},"train/train/tensor_act_model_layers_5_self_attn/norm":497.6350976881482,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_down_proj/mean":0.005405426025390625,"train/train/tensor_act_model_layers_4/std":0.2797865139362316,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1/max_abs":1.1015625,"train/train/tensor_act_model_layers_8_mlp_up_proj/mean":0.00576019287109375,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/mean":-4.833564162254333e-07,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/max_abs":0.00531005859375,"train/loss":49.74850769042969,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/std":0.0206298828125,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/norm":0.10246288558032977,"train/train/tensor_act_model_layers_1_post_attention_layernorm/std":1.0000054609987319,"train/train/tensor_act_model_layers_0_self_attn/max_abs":0.224609375,"train/train/tensor_act_model_layers_4_self_attn_q_proj/norm":8444.509770831344,"train/train/tensor_param_model_layers_7_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/mean":-2.4102628231048584e-06,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/mean":6.784684956073761e-06,"train/train/tensor_act_model_layers_0_post_attention_layernorm/max_abs":4.53125,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/std":0.0008139678337106606,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/max_abs":0.138671875,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/max_abs":6.21875,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/norm":4.53125,"train/train/tensor_act_model_layers_0_self_attn_o_proj/mean":-0.0005276203155517578,"train/train/tensor_act_model_layers_6_mlp_down_proj/mean":0.0057201385498046875,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/std":0.00015010444818410604,"train/train/tensor_grad_model_embed_tokens_weight/std":0.00048201806107111585,"train/train/tensor_act_model_layers_7_input_layernorm/max_abs":4.78125,"train/train/tensor_act_model_layers_0_post_attention_layernorm/norm":9158.038940433684,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/max_abs":0.005279541015625,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_o_proj/norm":486.0344744237919,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/max_abs":0.1728515625,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/std":0.044189453125,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_4/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/mean":0.00015926361083984375,"train/train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/max_abs":0.4375,"train/train/global/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/std":0.0013599450739407621,"train/train/tensor_act_model_layers_2_self_attn/norm":408.9549750343595,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/mean":-4.274479579180479e-07,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/mean":-7.200241088867188e-05,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/norm":0.9297782789516292,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/mean":-0.00016117095947265625,"train/learning_rate":0.0005,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/mean":0.0001068115234375,"train/train/tensor_act_model_layers_1_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/std":1.000000133579297,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_2_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_5/act/std":0.5537264622185337,"train/train/tensor_act_model_layers_1_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_7/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_down_proj/std":0.1015626140108337,"train/train/tensor_act_model_layers_8_self_attn/max_abs":0.8125,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/norm":0.6348835527475227,"train/train/tensor_act_model_layers_4_self_attn_k_proj/std":0.845461853783858,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/std":0.0004312305883395487,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/std":0.0235595703125,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/norm":0.5176637974496008,"train/train/tensor_act_model_layers_7_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_k_proj/mean":-0.012340545654296875,"train/train/tensor_act_model_layers_0_self_attn_v_proj/norm":1815.8166580376064,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model/mean":0.0232391357421875,"train/train/tensor_param_model_layers_3_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_4_input_layernorm/mean":0.03276824951171874,"train/train/tensor_act_model_rotary_emb/mean":0.333984375,"train/train/tensor_act_model_layers_4_self_attn_q_proj/std":0.9201686668813257,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/max_abs":0.076171875,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5/norm":2580.3944732393816,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/max_abs":0.0098876953125,"train/train/global/param/norm":75.44594149383137,"train/train/layer_model_layers_1/act/mean":0.005572208991417518,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/norm":0.5500951511265744,"train/train/tensor_act_model_layers_7_mlp/mean":-0.0019550323486328125,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/norm":5.09375,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/norm":0.07944339313752548,"train/train/tensor_act_model_layers_6_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp/std":0.11087058315602684,"train/train/tensor_act_model_layers_6_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/std":1.0000004451720916,"train/train/tensor_act_model_layers_8_self_attn_o_proj/norm":1078.046025070771,"train/train/tensor_act_model_rotary_emb/max_abs":1,"train/train/tensor_act_model_layers_7_mlp_up_proj/max_abs":1.1953125,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/max_abs":0.125,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/std":0.002488334277224233,"train/train/tensor_act_model_layers_6_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/mean":0.00019550323486328125,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/std":0.000623451731167683,"train/train/tensor_act_model_embed_tokens/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn/norm":486.0344744237919,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/std":0.0012711289102917448,"train/train/tensor_act_model_norm/norm":9158.885131843323,"train/grad_norm":25.125,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/mean":-9.822845458984375e-05,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/norm":0.4239274009773825,"train/train/layer_model_layers_0/grad/mean":8.951448334174666e-07,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4/max_abs":1.65625,"train/train/tensor_param_model_layers_6_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_8_mlp_down_proj/max_abs":1.1640625,"train/train/tensor_act_model_layers_7_mlp/max_abs":0.73046875,"train/train/layer_model_layers_4/act/mean":0.011836038185999943,"train/train/layer_model_layers_6/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/max_abs":0.08984375,"train/train/tensor_act_model_layers_4_self_attn_q_proj/max_abs":4.78125,"train/train/tensor_grad_model_embed_tokens_weight/max_abs":0.05859375,"train/train/tensor_act_model_layers_5_self_attn_q_proj/norm":8626.440264451365,"train/train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_6_mlp/std":0.11035170806644656,"train/train/tensor_act_model_layers_3/std":0.2701432434459629,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/norm":4.59375,"train/train/tensor_act_model_layers_7_mlp_down_proj/norm":1157.7220554563974,"train/train/tensor_act_model_layers_4_self_attn_v_proj/std":0.33703761500531826,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/max_abs":0.09423828125,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/norm":4.5,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/norm":4.21875,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/norm":0.30612391776426706,"train/train/tensor_act_model_norm/max_abs":4.75,"train/train/tensor_param_model_layers_1_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/max_abs":0.09033203125,"train/train/tensor_act_model_layers_6_input_layernorm/max_abs":4.65625,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/norm":0.1486808388473866,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/std":0.0004077178083436662,"train/train/tensor_act_model_layers_5_mlp/norm":929.9160394931031,"train/train/tensor_act_model_layers_2_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/mean":2.9325485229492188e-05,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/std":0.000429591868268785,"train/train/tensor_act_model_layers_7_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/mean":1.1696829460561277e-07,"train/train/tensor_act_model_layers_3_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_v_proj/max_abs":1.6328125,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_3/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_5/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/std":0.041259765625,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/mean":8.427305147051811e-07,"train/train/tensor_act_model_layers_3_self_attn/std":0.05308823489047627,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/std":0.0007551836297561284,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/std":0.041748046875,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/std":0.0238037109375,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/max_abs":0.0888671875,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/norm":4.71875,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/max_abs":0.0017547607421875,"train/train/tensor_act_model_layers_0_post_attention_layernorm/mean":0.0021638870239257812,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/mean":-0.000484466552734375,"train/train/tensor_act_model_layers_8/mean":0.008716583251953125,"train/train/tensor_act_model_layers_7_self_attn_q_proj/norm":10461.691153606722,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/mean":6.190501153469086e-06,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/norm":0.4110867073502657,"train/train/layer__model_layers_1/param/frac_near_user_limit":0,"train/train/tensor_param_model_embed_tokens_weight/norm":47.75,"train/train/layer_model_layers_4/act/max_abs":5.21875,"train/train/layer_model_layers_3/grad/norm":1.1688936654531,"train/train/tensor_act_model_layers_3_self_attn/mean":-0.00029712915420532227,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/mean":-3.6008714232593775e-07,"train/train/tensor_act_model_layers_5_mlp_up_proj/mean":-0.0006145238876342773,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/norm":4.84375,"train/train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/norm":0.057045541566442495,"train/train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_4/act/std":0.5473840823442112,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/max_abs":0.0849609375,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/std":0.02197265625,"train/train/tensor_act_model_layers_6_self_attn_q_proj/mean":-0.05267333984375,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_k_proj/norm":9821.311569001531,"train/train/tensor_act_model_layers_8_mlp_down_proj/mean":-0.0014677047729492188,"train/train/tensor_act_model_layers_1_mlp_up_proj/max_abs":1.2265625,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/max_abs":0.0018463134765625,"train/train/tensor_act_model_layers_7/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/norm":4.53125,"train/train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/norm":3.515625,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_embed_tokens/norm":686.0245066748595,"train/train/layer__model_layers_1/param/max_abs":1,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/norm":4.5,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/mean":4.982948303222656e-05,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/norm":3.21875,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/mean":-9.870529174804688e-05,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_2/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/max_abs":0.1328125,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/max_abs":0.001007080078125,"train/train/tensor_act_model_layers_2_mlp/norm":1050.6969802776157,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/mean":4.0802115108817816e-07,"train/train/layer__model_layers_4/param/norm":18.830361641010775,"train/train/tensor_act_/max_abs":3.1596059799194336,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/mean":-0.000225067138671875,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/max_abs":0.091796875,"train/train/tensor_act_model_layers_6_self_attn_v_proj/max_abs":2.328125,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/max_abs":0.0286865234375,"train/train/tensor_act_model_layers_8_self_attn_k_proj/mean":-0.0522308349609375,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/std":0.0218505859375,"train/train/tensor_act_model_layers_4/mean":0.007947921752929688,"train/train/tensor_param_model_embed_tokens_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/max_abs":0.0859375,"train/train/layer_model_layers_6/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp/mean":-0.0014677047729492188,"train/train/tensor_act_model_layers_3_self_attn_o_proj/mean":-0.00029712915420532227,"train/train/layer_model_layers_5/grad/std":0.0006830734147017943,"train/train/layer__model_layers_5/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_v_proj/norm":3463.3056672450866,"train/train/layer_model_layers_1/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_k_proj/max_abs":4.21875,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/norm":4.625,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_up_proj/std":0.1972049672413496,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/max_abs":0.009765625,"train/train/tensor_act_model_layers_7_input_layernorm/std":1.0000001769512734,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/norm":139.19229095220038,"train/train/tensor_act_model_layers_3_self_attn_o_proj/std":0.05308823489047627,"train/train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/norm":1.0784712525676086,"train/train/layer_model_layers_6/grad/mean":1.0562621859046114e-06,"train/train/layer_model_layers_1/grad/std":0.0010789055805118235,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_o_proj/max_abs":0.8125,"train/train/tensor_param_model_layers_4_input_layernorm_weight/std":0,"eval/runtime":9.5225,"train/train/tensor_act_model_layers_1_self_attn/mean":-0.0022516250610351562,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/norm":4.53125,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/global/grad/norm":4.998155692517693,"train/train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/mean":1.1994852684438229e-06,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/std":0.0004098031937883901,"train/train/tensor_act_model_layers_6_self_attn_v_proj/mean":-0.0036039352416992188,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/max_abs":0.080078125,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/std":0.0213623046875,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/mean":-4.38690185546875e-05,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/max_abs":0.125,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/std":0.00011141785897713587,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_8/grad/std":0.0005782672798617279,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/mean":-3.025401383638382e-06,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/max_abs":0.103515625,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/max_abs":0.0018768310546875,"train/train/tensor_act_model_layers_1_mlp_up_proj/norm":3503.9026158574925,"train/train/tensor_act_model_layers_3_self_attn_k_proj/mean":0.08953857421875,"train/train/tensor_act_model_layers_0_self_attn_o_proj/std":0.01518657306789068,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/global/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_v_proj/norm":3054.7713515068785,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/max_abs":0.1328125,"train/train/tensor_act_model_layers_7/norm":2730.395501469702,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/max_abs":0.0074462890625,"train/train/tensor_act_model_layers_8_self_attn_o_proj/mean":0.0038623809814453125,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_o_proj/std":0.08719287717656642,"train/train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/mean":-0.00015354156494140625,"train/train/tensor_act_model_layers_2_self_attn/mean":0.0020008087158203125,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/std":0.9414076213487585,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/std":0.00011609377149315461,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/max_abs":0.001678466796875,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/mean":-0.0002288818359375,"train/train/tensor_grad_model_embed_tokens_weight/mean":-6.477348506450653e-07,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_k_proj/std":0.8823257181255008,"train/train/total_time_seconds":676.0385475568473,"train/train/tensor_param_model_norm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/mean":1.9550323486328125e-05,"train/train/layer_model_layers_8/grad/norm":0.936803955234771,"train/train/tensor_act_model_layers_0_mlp_up_proj/std":0.19842569797190573,"train/train/tensor_param_model_layers_0_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_2/act/mean":0.008575276722415136,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_0_mlp_up_proj/norm":3148.9947247149603,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/std":0.001017107616730838,"train/train/tensor_act_model_layers_1_self_attn/norm":301.4580922175665,"train/train/tensor_act_lm_head/norm":137978.5112112536,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/max_abs":0.001800537109375,"train/train/tensor_grad_model_norm_weight/mean":-0.006618499755859375,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/max_abs":0.002593994140625,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/mean":-0.0001621246337890625,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_6/act/max_abs":6.03125,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/std":0.034912109375,"train/train/tensor_act_model_layers_7_input_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_3/param/norm":19.16069101254245,"train/train/layer__model_layers_5/param/norm":18.903182860714885,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/mean":2.3896805942058563e-05,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_v_proj/max_abs":1.40625,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/mean":-8.822977542877197e-05,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/mean":0.0002689361572265625,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/mean":-8.440017700195312e-05,"_runtime":1557,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/std":0.039306640625,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/max_abs":0.16796875,"train/train/tensor_act_model_layers_3_input_layernorm/norm":9158.836547869432,"train/train/tensor_act_model_layers_7_mlp_up_proj/mean":-0.0019998550415039067,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_0/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/std":0.32312162517924214,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/std":0.032958984375,"train/train/layer__model_layers_6/param/std":0.047603568559157615,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/norm":3.171875,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/std":0.0007029179815078741,"train/train/tensor_act_model_layers_0_self_attn_k_proj/max_abs":4.3125,"train/train/layer_model_layers_7/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/mean":4.7208741307258614e-06,"train/train/layer_model_layers_4/grad/max_abs":0.00689697265625,"train/train/tensor_grad_model_norm_weight/max_abs":0.01385498046875,"train/train/tensor_act_model_layers_2_mlp_up_proj/max_abs":1.1484375,"train/train/layer__model_layers_5/param/std":0.046641797054845655,"train/train/layer__model_layers_4/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_input_layernorm/norm":9158.866333015456,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_q_proj/norm":7272.671681066942,"train/train/tensor_act_model_layers_8_mlp/std":0.20031891610751495,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/std":0.0002903939882593242,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/max_abs":0.0019378662109375,"train/train/layer_model_layers_4/grad/norm":1.0366429293152521,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/norm":0.569249665150902,"train/train/tensor_act_model_layers_3_post_attention_layernorm/std":1.0000005670588714,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/std":0.00018768759095130736,"train/train/tensor_act_model_layers_7_self_attn/max_abs":0.8203125,"train/train/tensor_act_model_layers_3_mlp/norm":998.9542597675202,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_0/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/max_abs":0.1484375,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/std":0.000614104388010392,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/max_abs":0.00115203857421875,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/norm":0.08442042505521935,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_o_proj/std":0.03286092571977793,"train/train/tensor_act_model_layers_4_post_attention_layernorm/max_abs":5.21875,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/mean":-1.043081283569336e-06,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn/mean":-0.0008254051208496094,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_6/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/std":0.0230712890625,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/mean":-2.4600012693554163e-08,"train/train/tensor_act_model_layers_1_input_layernorm/std":1.0000012806912975,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/norm":0.018508678412802338,"train/train/tensor_act_model_layers_8_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/grad/std":0.0007216427637524477,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_6_mlp/mean":0.0057201385498046875,"train/train/layer_model_layers_1/act/max_abs":5.5,"train/train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_o_proj/std":0.0446176132967603,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_input_layernorm/max_abs":4.65625,"_step":53,"train/train/tensor_act_model_layers_3_mlp/mean":-0.0022554397583007812,"train/train/tensor_act_model_layers_4_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn/std":0.03286092571977793,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/norm":0.10740945917684312,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/norm":0.2792116048333057,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/max_abs":0.0015716552734375,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/mean":0.0004425048828125,"train/train/tensor_act_model_layers_5_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_2/frac_near_dtype_limit":0,"train/train/tensor_param_model_embed_tokens_weight/mean":-0.00093841552734375,"train/train/tensor_act_model_layers_1_mlp/mean":0.0067119598388671875,"train/train/layer__model_layers_5/param/max_abs":1,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/std":0.036865234375,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/mean":1.3531680451706052e-06,"train/train/tensor_act_model_layers_6_mlp_down_proj/std":0.11035170806644656,"train/train/layer_model_layers_6/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/max_abs":0.73046875,"train/train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_user_limit":0,"train/train/global/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp_up_proj/std":0.2207643275831124,"train/train/tensor_act_model_layers_5_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_2/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/std":1.0625039114311987,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/norm":2.796875,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/norm":0.5447748457558951,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/max_abs":0.09765625,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/grad/max_abs":0.00823974609375,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/mean":-5.647540092468262e-05,"train/train/tensor_param_model_layers_0_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_6/norm":2675.436665474263,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/max_abs":0.10546875,"train/train/tensor_act_model_layers_5_input_layernorm/max_abs":4.84375,"train/train/tensor_act_model_layers_2_self_attn_k_proj/std":0.8444851065697362,"train/train/layer__model_layers_6/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2/std":0.24310367211460596,"train/train/tensor_act_model_rotary_emb/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn/frac_near_dtype_limit":0,"train/train/layer__model_layers_7/param/norm":19.057209819219338,"train/train/layer__model_layers_6/param/max_abs":1,"train/train/tensor_act_model_layers_8_self_attn_k_proj/max_abs":6.375,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1/norm":1821.4157893504037,"train/train/tensor_act_model/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6/max_abs":1.6640625,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/mean":-2.944469451904297e-05,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/max_abs":0.004913330078125,"train/train/tensor_act_model_layers_3_self_attn_v_proj/std":0.3339849269576031,"train/train/tensor_act_model_layers_4_self_attn_o_proj/max_abs":0.60546875,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/std":0.0005703787486938851,"train/train/tensor_act_model_layers_4_post_attention_layernorm/std":1.000000227126267,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/max_abs":0.01483154296875,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_7_self_attn/norm":870.6736716013772,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/std":0.0361328125,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp/std":0.10058596968487546,"train/train/tensor_act_model_layers_0_mlp_down_proj/mean":9.156763553619385e-06,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/std":0.00046367620843929094,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/std":0.00017697632508598447,"train/train/layer_model_layers_7/grad/std":0.0006054649694080247,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/std":0.0006207572805987063,"train/train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/norm":2.206872228784369,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/norm":0.7576419599101569,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/norm":0.0647402350765535,"train/train/layer_model_layers_1/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_k_proj/std":0.8195828502245655,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/std":0.0228271484375,"train/train/tensor_act_model_layers_3_mlp/std":0.10897847648259888,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_4_input_layernorm/max_abs":4.8125,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/mean":0.0002346038818359375,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/max_abs":0.0111083984375,"train/train/tensor_act_model_layers_1_self_attn_q_proj/std":0.7934583210567848,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/max_abs":0.083984375,"train/train/tensor_act_model_layers_5_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_4_mlp_up_proj/norm":3015.185914880472,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/mean":6.67572021484375e-05,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_norm_weight/max_abs":1,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/norm":4.59375,"train/train/tensor_act_model_layers_7_post_attention_layernorm/norm":9158.870239264383,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0/norm":1072.8987485266305,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_v_proj/std":0.36547945254426323,"train/train/tensor_act_model_layers_8_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/mean":-4.631292540580035e-06,"train/train/tensor_act_/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/mean":7.437483873218298e-07,"train/train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_2/param/mean":0.0015293722405634506,"train/train/tensor_act_model_layers_5_mlp_down_proj/norm":929.9160394931031,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/norm":3.0625,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/mean":1.424551010131836e-05,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/std":0.198242266075022,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/mean":-1.1563301086425781e-05,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_0_self_attn_q_proj/mean":-0.017574310302734375,"train/train/tensor_act_model_layers_0_self_attn/norm":139.19229095220038,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/max_abs":0.007171630859375,"train/train/tensor_act_model_layers_8_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp/norm":1157.7220554563974,"train/train/tensor_act_lm_head/std":1.6914375425620019,"train/train/tensor_param_model_layers_1_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_8_mlp/norm":1833.4078311255546,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/std":0.02392578125,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/max_abs":0.001495361328125,"train/train/tensor_act_model_layers_3_self_attn_v_proj/mean":-0.002701282501220703,"train/train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/norm":0.0728063202556278,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/std":0.00020978625517021878,"train/train/tensor_act_model_layers_0_self_attn_k_proj/norm":7571.323294073769,"train/train/tensor_act_model_embed_tokens/max_abs":0.25,"train/train/tensor_act_model_layers_3_mlp_up_proj/mean":-0.0009462833404541016,"train/train/tensor_act_model_layers_1_self_attn_v_proj/std":0.271851848590517,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/mean":6.760121323168278e-06,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/max_abs":0.0022735595703125,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/std":0.0018718482317452295,"train/train/tensor_act_model_layers_0_input_layernorm/max_abs":3.71875,"train/train/tensor_act_model_layers_1/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_input_layernorm/std":1.0000012683441275,"train/train/tensor_act_model_layers_1/std":0.1989142884855857,"train/train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/norm":0.5091221281177153,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/norm":0.05909136233456057,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/std":0.0016695101003668653,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/std":0.037841796875,"train/train/layer__model_layers_4/param/max_abs":1,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/max_abs":0.09814453125,"train/train/tensor_act_model_layers_1_mlp_down_proj/mean":0.0067119598388671875,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/mean":7.352093234658241e-06,"train/train/tensor_act_model_layers_0/frac_near_dtype_limit":0,"train/train/layer_model_layers_1/act/norm":17389.814008941397,"train/train/tensor_act_model_layers_8_post_attention_layernorm/std":1.0000002402811954,"train/train/tensor_act_model_layers_5_self_attn_o_proj/std":0.05436818972384846,"train/train/tensor_act_model_layers_7_self_attn_v_proj/max_abs":2.046875,"train/train/layer_model_layers_8/act/norm":22737.438562436826,"train/train/tensor_act_model_layers_6_self_attn_k_proj/std":1.068372004136195,"train/train/tensor_act_model_layers_3/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/act/max_abs":6.21875,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/mean":1.557054929435253e-08,"train/train/tensor_act_model_layers_0_self_attn_q_proj/std":0.8996636131643495,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_post_attention_layernorm/mean":0.02803802490234375,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/std":0.0003992499171670238,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/mean":5.144684109836817e-07,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/std":0.0001551661666112421,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/std":0.0008548833918412956} \ No newline at end of file diff --git a/wandb/run-20260809_055344-aqwnomdl/logs/debug-core.log b/wandb/run-20260809_055344-aqwnomdl/logs/debug-core.log new file mode 100644 index 0000000000000000000000000000000000000000..26b98ec4487451545c2fa62bcc55b922ef41ffd2 --- /dev/null +++ b/wandb/run-20260809_055344-aqwnomdl/logs/debug-core.log @@ -0,0 +1,20 @@ +{"time":"2026-08-09T05:53:44.18492886Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmpz7t0nmko/port-4021702.txt","pid":4021702,"detached":false,"idle-timeout":600000000000,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false} +{"time":"2026-08-09T05:53:44.185414223Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":4021702} +{"time":"2026-08-09T05:53:44.185377244Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-4021702-4022833-3011194613/socket","Net":"unix"}} +{"time":"2026-08-09T05:53:44.36271518Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"} +{"time":"2026-08-09T05:53:44.450053111Z","level":"INFO","msg":"handleInformInit: received","streamId":"aqwnomdl","id":"1(@)"} +{"time":"2026-08-09T05:53:44.709794362Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"aqwnomdl","id":"1(@)"} +{"time":"2026-08-09T05:53:50.029649939Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"yhcgdr9or0xm"} +{"time":"2026-08-09T06:19:42.045735696Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"yhcgdr9or0xm"} +{"time":"2026-08-09T06:19:42.607881575Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"yhcgdr9or0xm"} +{"time":"2026-08-09T06:19:42.612182914Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"aqwnomdl","id":"1(@)"} +{"time":"2026-08-09T06:19:42.612800411Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"aqwnomdl","id":"1(@)"} +{"time":"2026-08-09T06:19:42.827218828Z","level":"INFO","msg":"handleInformTeardown: server teardown initiated","id":"1(@)"} +{"time":"2026-08-09T06:19:42.827273806Z","level":"INFO","msg":"handleInformTeardown: server shutdown complete","id":"1(@)"} +{"time":"2026-08-09T06:19:42.82728264Z","level":"INFO","msg":"connection: closing","id":"1(@)"} +{"time":"2026-08-09T06:19:42.827300478Z","level":"INFO","msg":"server: is shutting down"} +{"time":"2026-08-09T06:19:42.827307947Z","level":"INFO","msg":"processOutgoingData: finished","id":"1(@)"} +{"time":"2026-08-09T06:19:42.827330894Z","level":"INFO","msg":"connection: closed successfully","id":"1(@)"} +{"time":"2026-08-09T06:19:42.827421899Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"1(@)"} +{"time":"2026-08-09T06:19:42.827496775Z","level":"INFO","msg":"server: listener closed","addr":{"Name":"/tmp/wandb-4021702-4022833-3011194613/socket","Net":"unix"}} +{"time":"2026-08-09T06:19:42.827561911Z","level":"INFO","msg":"server: all connections closed"} diff --git a/wandb/run-20260809_055344-aqwnomdl/logs/debug-internal.log b/wandb/run-20260809_055344-aqwnomdl/logs/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..ffb53c89804d8f54e073062ad637fb3a4244003f --- /dev/null +++ b/wandb/run-20260809_055344-aqwnomdl/logs/debug-internal.log @@ -0,0 +1,225 @@ +{"time":"2026-08-09T05:53:44.450225204Z","level":"INFO","msg":"wandb-core"} +{"time":"2026-08-09T05:53:44.450523278Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"} +{"time":"2026-08-09T05:53:44.709606557Z","level":"INFO","msg":"stream: created new stream","id":"aqwnomdl"} +{"time":"2026-08-09T05:53:44.709699351Z","level":"INFO","msg":"handler: started"} +{"time":"2026-08-09T05:53:44.709783769Z","level":"INFO","msg":"stream: started"} +{"time":"2026-08-09T05:53:44.709808909Z","level":"INFO","msg":"writer: started","stream_id":"aqwnomdl"} +{"time":"2026-08-09T05:53:44.709823807Z","level":"INFO","msg":"sender: started"} +{"time":"2026-08-09T05:53:46.170317403Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1} +{"time":"2026-08-09T05:53:46.266265258Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:54:01.171241322Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":0,"events_lines":2,"console_offset":1,"console_lines":4,"uploaded_len":2} +{"time":"2026-08-09T05:54:01.272715296Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:54:16.171007283Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":2,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:54:16.258424558Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:54:27.910615844Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":81} +{"time":"2026-08-09T05:54:27.942692362Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1678} +{"time":"2026-08-09T05:54:31.175114614Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":1,"events_offset":4,"events_lines":2,"console_offset":4,"console_lines":2} +{"time":"2026-08-09T05:54:31.351312086Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:54:46.170516928Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":6,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:54:46.267246601Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:55:01.171200078Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":8,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:55:01.254117166Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:55:16.172465012Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":1,"history_lines":1,"events_offset":10,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:55:16.341305976Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:55:31.17077432Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":12,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:55:31.264475304Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:55:46.172188369Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":2,"history_lines":1,"events_offset":14,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:55:46.325141167Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:56:01.172627163Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":3,"history_lines":1,"events_offset":16,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:56:01.346524784Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:56:16.171074771Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":18,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:56:16.264056406Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:56:31.172612865Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":4,"history_lines":1,"events_offset":20,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:56:31.331132034Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:56:46.171395807Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":22,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:56:46.26192667Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:57:01.170553491Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":24,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:57:01.256324797Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:57:16.172719704Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":5,"history_lines":2,"events_offset":26,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:57:16.310119871Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:57:31.171076856Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":28,"events_lines":2,"console_offset":6,"console_lines":11} +{"time":"2026-08-09T05:57:31.272630812Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:57:46.172423254Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":7,"history_lines":1,"events_offset":30,"events_lines":2,"console_offset":16,"console_lines":2} +{"time":"2026-08-09T05:57:46.314453888Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:58:01.170946727Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":32,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:58:01.262194219Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:58:16.171278605Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":34,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:58:16.259157915Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:58:31.172573636Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":8,"history_lines":1,"events_offset":36,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:58:31.317489099Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:58:46.170440342Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":38,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:58:46.243173989Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:59:01.172646214Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":9,"history_lines":1,"events_offset":40,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:59:01.319550498Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:59:16.172574642Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":10,"history_lines":1,"events_offset":42,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:59:16.328666295Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:59:31.171127468Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":44,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:59:31.263499672Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:59:46.176023892Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":11,"history_lines":1,"events_offset":46,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T05:59:46.342768655Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:00:01.17132454Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":48,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:00:01.264426451Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:00:16.170670657Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":50,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:00:16.256441947Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:00:31.172081894Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":12,"history_lines":1,"events_offset":52,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:00:31.341987778Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:00:46.172863646Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":13,"history_lines":1,"events_offset":54,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:00:46.353507775Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:01:01.171246197Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":56,"events_lines":2,"console_offset":18,"console_lines":11} +{"time":"2026-08-09T06:01:01.308787662Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:01:16.171885054Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":14,"history_lines":1,"events_offset":58,"events_lines":2,"console_offset":28,"console_lines":2} +{"time":"2026-08-09T06:01:16.324264802Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:01:31.171014258Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":60,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:01:31.252784524Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:01:46.170678501Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":62,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:01:46.261392297Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:02:01.172164341Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":15,"history_lines":1,"events_offset":64,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:02:01.336523754Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:02:16.172406277Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":16,"history_lines":1,"events_offset":66,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:02:16.34581226Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:02:31.17114336Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":68,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:02:31.264728149Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:02:46.17251556Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":17,"history_lines":1,"events_offset":70,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:02:46.356565867Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:03:01.171050414Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":72,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:03:01.272046207Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:03:16.172259018Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":18,"history_lines":1,"events_offset":74,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:03:16.376295656Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:03:31.170699734Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":76,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:03:31.25223725Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:03:46.170517552Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":78,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:03:46.269644144Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:04:01.172566349Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":19,"history_lines":2,"events_offset":80,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:04:01.395308985Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:04:16.170670162Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":82,"events_lines":2,"console_offset":30,"console_lines":11} +{"time":"2026-08-09T06:04:16.259507833Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:04:31.171280096Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":84,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:04:31.274626369Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:04:46.171884086Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":21,"history_lines":1,"events_offset":86,"events_lines":2,"console_offset":40,"console_lines":2} +{"time":"2026-08-09T06:04:46.404744761Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:05:01.170690641Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":88,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:05:01.271330338Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:05:16.172350628Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":22,"history_lines":1,"events_offset":90,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:05:16.318301465Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:05:31.170854864Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":92,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:05:31.270064132Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:05:46.174023841Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":23,"history_lines":1,"events_offset":94,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:05:46.31610268Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:06:01.172282041Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":24,"history_lines":1,"events_offset":96,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:06:01.325276904Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:06:16.170418883Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":98,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:06:16.270554146Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:06:31.171070078Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":100,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:06:31.258742671Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:06:46.172734022Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":25,"history_lines":1,"events_offset":102,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:06:46.330272981Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:07:01.170603193Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":104,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:07:01.2916019Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:07:16.172177282Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":26,"history_lines":1,"events_offset":106,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:07:16.320893827Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:07:31.172796791Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":27,"history_lines":1,"events_offset":108,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:07:31.318187341Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:07:46.170764874Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":110,"events_lines":2,"console_offset":42,"console_lines":11} +{"time":"2026-08-09T06:07:46.257368025Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:08:01.172356584Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":28,"history_lines":1,"events_offset":112,"events_lines":2,"console_offset":52,"console_lines":2} +{"time":"2026-08-09T06:08:01.318136202Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:08:16.171163119Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":114,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:08:16.251006636Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:08:31.171250802Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":116,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:08:31.261666498Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:08:46.172793077Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":29,"history_lines":1,"events_offset":118,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:08:46.314737202Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:09:01.170351226Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":120,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:09:01.265428095Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:09:16.172414877Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":30,"history_lines":1,"events_offset":122,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:09:16.334815012Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:09:31.171832894Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":31,"history_lines":1,"events_offset":124,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:09:31.315486921Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:09:46.170582645Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":126,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:09:46.269033463Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:10:01.170632924Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":128,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:10:01.262808609Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:10:16.172991321Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":32,"history_lines":1,"events_offset":130,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:10:16.338084996Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:10:31.171073448Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":132,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:10:31.271284887Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:10:46.171995962Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":33,"history_lines":1,"events_offset":134,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:10:46.343864332Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:11:01.172562883Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":34,"history_lines":1,"events_offset":136,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:11:01.33419194Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:11:16.171203228Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":138,"events_lines":2,"console_offset":54,"console_lines":11} +{"time":"2026-08-09T06:11:16.26276551Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:11:31.172065579Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":35,"history_lines":1,"events_offset":140,"events_lines":2,"console_offset":64,"console_lines":2} +{"time":"2026-08-09T06:11:31.338683054Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:11:46.170659494Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":142,"events_lines":2,"console_offset":64,"console_lines":1} +{"time":"2026-08-09T06:11:46.267239271Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:12:01.171145736Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":144,"events_lines":2,"console_offset":64,"console_lines":1} +{"time":"2026-08-09T06:12:01.264702722Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:12:16.171836693Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":36,"history_lines":1,"events_offset":146,"events_lines":2,"console_offset":64,"console_lines":1} +{"time":"2026-08-09T06:12:16.312813468Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:12:31.170510141Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":148,"events_lines":2,"console_offset":64,"console_lines":1} +{"time":"2026-08-09T06:12:31.256345605Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:12:46.172873827Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":37,"history_lines":1,"events_offset":150,"events_lines":2,"console_offset":64,"console_lines":1} +{"time":"2026-08-09T06:12:46.320410476Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:13:01.172684854Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":38,"history_lines":1,"events_offset":152,"events_lines":2,"console_offset":64,"console_lines":1} +{"time":"2026-08-09T06:13:01.363377259Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:13:16.171224999Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":154,"events_lines":2,"console_offset":64,"console_lines":1} +{"time":"2026-08-09T06:13:16.25965355Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:13:31.171933703Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":39,"history_lines":1,"events_offset":156,"events_lines":2,"console_offset":64,"console_lines":1} +{"time":"2026-08-09T06:13:31.339119368Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:13:46.171013247Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":158,"events_lines":2,"console_offset":64,"console_lines":1} +{"time":"2026-08-09T06:13:46.25534231Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:14:01.171081133Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":160,"events_lines":2,"console_offset":64,"console_lines":1} +{"time":"2026-08-09T06:14:01.271329894Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:14:16.172401591Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":40,"history_lines":1,"events_offset":162,"events_lines":2,"console_offset":64,"console_lines":1} +{"time":"2026-08-09T06:14:16.31550435Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:14:31.172167132Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":41,"history_lines":1,"events_offset":164,"events_lines":2,"console_offset":64,"console_lines":1} +{"time":"2026-08-09T06:14:31.325500427Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:14:46.170486083Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":166,"events_lines":2,"console_offset":66,"console_lines":11} +{"time":"2026-08-09T06:14:46.255707598Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:15:01.172338355Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":42,"history_lines":1,"events_offset":168,"events_lines":2,"console_offset":76,"console_lines":2} +{"time":"2026-08-09T06:15:01.3397005Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:15:16.170369609Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":170,"events_lines":2,"console_offset":76,"console_lines":1} +{"time":"2026-08-09T06:15:16.253343016Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:15:31.172365221Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":43,"history_lines":1,"events_offset":172,"events_lines":2,"console_offset":76,"console_lines":1} +{"time":"2026-08-09T06:15:31.346131863Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:15:46.171103087Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":174,"events_lines":2,"console_offset":76,"console_lines":1} +{"time":"2026-08-09T06:15:46.250662989Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:16:01.174102312Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":44,"history_lines":1,"events_offset":176,"events_lines":2,"console_offset":76,"console_lines":1} +{"time":"2026-08-09T06:16:01.354597224Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:16:16.170603597Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":178,"events_lines":2,"console_offset":76,"console_lines":1} +{"time":"2026-08-09T06:16:16.249994507Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:16:31.171889015Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":45,"history_lines":1,"events_offset":180,"events_lines":2,"console_offset":76,"console_lines":1} +{"time":"2026-08-09T06:16:31.354842703Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:16:46.170754981Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":182,"events_lines":2,"console_offset":76,"console_lines":1} +{"time":"2026-08-09T06:16:46.269136337Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:17:01.172610358Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":46,"history_lines":1,"events_offset":184,"events_lines":2,"console_offset":76,"console_lines":1} +{"time":"2026-08-09T06:17:01.348692717Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:17:16.170929835Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":186,"events_lines":2,"console_offset":76,"console_lines":1} +{"time":"2026-08-09T06:17:16.27437682Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:17:31.170813497Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":188,"events_lines":2,"console_offset":76,"console_lines":1} +{"time":"2026-08-09T06:17:31.26344185Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:17:46.172468576Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":47,"history_lines":2,"events_offset":190,"events_lines":2,"console_offset":76,"console_lines":1} +{"time":"2026-08-09T06:17:46.341784438Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:18:01.170601655Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":192,"events_lines":2,"console_offset":78,"console_lines":11} +{"time":"2026-08-09T06:18:01.256789367Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:18:16.17109947Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":194,"events_lines":2,"console_offset":88,"console_lines":1} +{"time":"2026-08-09T06:18:16.265868853Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:18:31.173724068Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":49,"history_lines":1,"events_offset":196,"events_lines":2,"console_offset":88,"console_lines":2} +{"time":"2026-08-09T06:18:31.32813923Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:18:46.170681197Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":198,"events_lines":2,"console_offset":88,"console_lines":1} +{"time":"2026-08-09T06:18:46.272342513Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:19:01.172044885Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":50,"history_lines":1,"events_offset":200,"events_lines":2,"console_offset":88,"console_lines":1} +{"time":"2026-08-09T06:19:01.319263314Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:19:16.170669116Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":202,"events_lines":2,"console_offset":90,"console_lines":3} +{"time":"2026-08-09T06:19:16.249013118Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:19:31.174940613Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":51,"history_lines":2,"events_offset":204,"events_lines":2,"console_offset":92,"console_lines":8} +{"time":"2026-08-09T06:19:31.384800165Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:19:42.451478007Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"} +{"time":"2026-08-09T06:19:42.455189198Z","level":"INFO","msg":"filestream: sending request","total_files":3,"history_offset":53,"history_lines":1,"console_offset":100,"console_lines":28,"uploaded_len":3,"complete":true,"exit_code":0} +{"time":"2026-08-09T06:19:42.602397475Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:19:42.60406777Z","level":"INFO","msg":"handler: operation stats","stats":{}} +{"time":"2026-08-09T06:19:42.61221407Z","level":"INFO","msg":"stream: finishing up"} +{"time":"2026-08-09T06:19:42.612232143Z","level":"INFO","msg":"handler: closed"} +{"time":"2026-08-09T06:19:42.612342712Z","level":"INFO","msg":"sender: closed"} +{"time":"2026-08-09T06:19:42.612348098Z","level":"INFO","msg":"stream: all finished"} diff --git a/wandb/run-20260809_055344-aqwnomdl/logs/debug.log b/wandb/run-20260809_055344-aqwnomdl/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..947aa1069e81923afb8d7ae79ef3834551df024e --- /dev/null +++ b/wandb/run-20260809_055344-aqwnomdl/logs/debug.log @@ -0,0 +1,27 @@ +2026-08-09 05:53:44,448 INFO MainThread:4021702 [wandb_setup.py:_flush():81] Current SDK version is 0.28.1 +2026-08-09 05:53:44,448 INFO MainThread:4021702 [wandb_setup.py:_flush():81] Configure stats pid to 4021702 +2026-08-09 05:53:44,448 INFO MainThread:4021702 [wandb_setup.py:_flush():81] Loading settings from environment variables +2026-08-09 05:53:44,448 INFO MainThread:4021702 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_055344-aqwnomdl/logs/debug.log +2026-08-09 05:53:44,448 INFO MainThread:4021702 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_055344-aqwnomdl/logs/debug-internal.log +2026-08-09 05:53:44,448 INFO MainThread:4021702 [wandb_init.py:init():772] calling init triggers +2026-08-09 05:53:44,448 INFO MainThread:4021702 [wandb_init.py:init():777] wandb.init called with sweep_config: {} +config: {'_wandb': {}} +2026-08-09 05:53:44,448 INFO MainThread:4021702 [wandb_init.py:init():820] starting backend +2026-08-09 05:53:44,448 INFO MainThread:4021702 [wandb_init.py:init():835] sending inform_init request +2026-08-09 05:53:44,710 INFO MainThread:4021702 [wandb_init.py:init():840] backend started and connected +2026-08-09 05:53:44,711 INFO MainThread:4021702 [wandb_init.py:init():910] updated telemetry +2026-08-09 05:53:44,718 INFO MainThread:4021702 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout +2026-08-09 05:53:44,952 INFO MainThread:4021702 [wandb_init.py:init():978] starting run threads in backend +2026-08-09 05:53:45,026 INFO MainThread:4021702 [wandb_run.py:_console_start():2621] atexit reg +2026-08-09 05:53:45,026 INFO MainThread:4021702 [wandb_run.py:_redirect():2471] redirect: wrap_raw +2026-08-09 05:53:45,026 INFO MainThread:4021702 [wandb_run.py:_redirect():2540] Wrapping output streams. +2026-08-09 05:53:45,026 INFO MainThread:4021702 [wandb_run.py:_redirect():2563] Redirects installed. +2026-08-09 05:53:45,029 INFO MainThread:4021702 [wandb_init.py:init():1016] run started, returning control to user process +2026-08-09 05:53:45,030 INFO MainThread:4021702 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.15.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 9, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 'linear', 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'outio/mlp-linear-9L_run', 'per_device_train_batch_size': 80, 'num_train_epochs': 1, 'max_steps': 750, 'learning_rate': 0.0005, 'lr_scheduler_type': 'constant', 'lr_scheduler_kwargs': None, 'warmup_steps': 0, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.01, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 16, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-linear-9L-2.0M-20260809-055343', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 50, 'eval_delay': 0, 'per_device_eval_batch_size': 80, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': True, 'hub_token': '', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/finale-mlp-linear-9L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1} +2026-08-09 05:53:45,032 INFO MainThread:4021702 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 2001280 - > +2026-08-09 05:53:45,032 INFO MainThread:4021702 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 2001280 None +2026-08-09 06:19:42,044 INFO MainThread:4021702 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/huggingface/aqwnomdl +2026-08-09 06:19:42,045 INFO MainThread:4021702 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0 +2026-08-09 06:19:42,045 INFO MainThread:4021702 [wandb_run.py:_restore():2570] restore +2026-08-09 06:19:42,045 INFO MainThread:4021702 [wandb_run.py:_restore():2576] restore done +2026-08-09 06:19:42,611 INFO MainThread:4021702 [wandb_run.py:_footer_sync_info():3993] logging synced files diff --git a/wandb/run-20260809_055344-aqwnomdl/run-aqwnomdl.wandb b/wandb/run-20260809_055344-aqwnomdl/run-aqwnomdl.wandb new file mode 100644 index 0000000000000000000000000000000000000000..73cb2eae4a47a22968cc913ccb469ece5a56b88a --- /dev/null +++ b/wandb/run-20260809_055344-aqwnomdl/run-aqwnomdl.wandb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b41e4000f3f973e311bd82a22f1b02fc9304e04ea24cbbde704a6ec28f49e877 +size 2187702 diff --git a/wandb/run-20260809_055726-m1dnjnh6/files/config.yaml b/wandb/run-20260809_055726-m1dnjnh6/files/config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..0a8bd6183d6694b7f6dc316ce55cb7c470b6aa26 --- /dev/null +++ b/wandb/run-20260809_055726-m1dnjnh6/files/config.yaml @@ -0,0 +1,433 @@ +_name_or_path: + value: "" +_wandb: + value: + cli_version: 0.28.1 + e: + k97nc6zhnp2eno8n59e5ngav280824l8: + args: + - --config + - configs/baseline.yaml + - --variants + - glu-waleed-94L + - --push + codePath: sweep.py + codePathLocal: sweep.py + cpu_count: 112 + cpu_count_logical: 224 + cudaVersion: "12.4" + disk: + /: + total: "1560765693952" + used: "708276572160" + email: deepnevro@gmail.com + executable: /mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python + git: + commit: 34b8d2e8f9a0c5751333310e69fa0c1056381deb + remote: https://github.com/deepnevro/Activation.git + gpu: NVIDIA H100 80GB HBM3 + gpu_count: 8 + gpu_nvidia: + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-39c684a5-fde6-83d7-1663-0859795881ae + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-132944c4-b689-2b5f-89a4-d730401677ab + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bc6c3e3c-9b90-09ca-c034-774961847c54 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea + host: deeplens-k3s-node1 + memory: + total: "2164089937920" + os: Linux-5.15.0-126-generic-x86_64-with-glibc2.35 + program: /mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py + python: CPython 3.11.15 + root: /mnt/data/zainulabideen/zain-exp/notebooks/Activation + startedAt: "2026-08-09T05:57:26.362395Z" + writerId: k97nc6zhnp2eno8n59e5ngav280824l8 + m: + - "1": train/global_step + "6": + - 3 + "7": [] + - "2": '*' + "5": 1 + "6": + - 1 + "7": [] + python_version: 3.11.15 + t: + "1": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "2": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "3": + - 2 + - 7 + - 13 + - 19 + - 41 + - 62 + - 66 + "4": 3.11.15 + "5": 0.28.1 + "6": 5.15.0.dev0 + "9": + "1": transformers_trainer + "12": 0.28.1 + "13": linux-x86_64 +accelerator_config: + value: + dispatch_batches: null + even_batches: true + gradient_accumulation_kwargs: null + non_blocking: false + split_batches: false + use_seedable_sampler: true +activation: + value: waleed +adam_beta1: + value: 0.9 +adam_beta2: + value: 0.999 +adam_epsilon: + value: 1e-08 +architectures: + value: null +attention_bias: + value: false +attention_dropout: + value: 0 +auto_find_batch_size: + value: false +average_tokens_across_devices: + value: true +batch_eval_metrics: + value: false +bf16: + value: true +bf16_full_eval: + value: false +bos_token_id: + value: 1 +chunk_size_feed_forward: + value: 0 +data_seed: + value: 42 +dataloader_drop_last: + value: false +dataloader_in_order: + value: true +dataloader_multiprocessing_context: + value: null +dataloader_num_workers: + value: 0 +dataloader_persistent_workers: + value: false +dataloader_pin_memory: + value: true +dataloader_prefetch_factor: + value: null +ddp_backend: + value: null +ddp_broadcast_buffers: + value: null +ddp_bucket_cap_mb: + value: null +ddp_find_unused_parameters: + value: null +ddp_static_graph: + value: null +ddp_timeout: + value: 1800 +debug: + value: [] +deepspeed: + value: null +disable_tqdm: + value: false +do_eval: + value: true +do_predict: + value: false +do_train: + value: false +dtype: + value: null +enable_jit_checkpoint: + value: false +eos_token_id: + value: 2 +eval_accumulation_steps: + value: null +eval_delay: + value: 0 +eval_do_concat_batches: + value: true +eval_on_start: + value: false +eval_steps: + value: 50 +eval_strategy: + value: steps +eval_use_gather_object: + value: false +fp16: + value: false +fp16_full_eval: + value: false +fsdp: + value: null +fsdp_config: + value: null +full_determinism: + value: false +gradient_accumulation_steps: + value: 4 +gradient_checkpointing: + value: false +gradient_checkpointing_kwargs: + value: null +greater_is_better: + value: null +head_dim: + value: 32 +hidden_act: + value: silu +hidden_size: + value: 128 +hub_always_push: + value: false +hub_model_id: + value: w-ahmad/A-glu-waleed-94L +hub_private_repo: + value: null +hub_revision: + value: null +hub_strategy: + value: every_save +hub_token: + value: +id2label: + value: + "0": LABEL_0 + "1": LABEL_1 +ignore_data_skip: + value: false +include_for_metrics: + value: [] +include_num_input_tokens_seen: + value: "no" +initializer_range: + value: 0.02 +intermediate_size: + value: 256 +is_encoder_decoder: + value: false +label_names: + value: null +label_smoothing_factor: + value: 0 +label2id: + value: + LABEL_0: 0 + LABEL_1: 1 +learning_rate: + value: 0.001 +length_column_name: + value: length +liger_kernel_config: + value: null +load_best_model_at_end: + value: false +local_rank: + value: -1 +log_level: + value: passive +log_level_replica: + value: warning +log_on_each_node: + value: true +logging_first_step: + value: false +logging_nan_inf_filter: + value: true +logging_steps: + value: 20 +logging_strategy: + value: steps +lr_scheduler_kwargs: + value: null +lr_scheduler_type: + value: constant +max_grad_norm: + value: 1 +max_position_embeddings: + value: 512 +max_steps: + value: 1500 +metric_for_best_model: + value: null +mlp_bias: + value: false +mlp_type: + value: glu +model/num_parameters: + value: 15949440 +model_type: + value: tiny_llama +neftune_noise_alpha: + value: null +num_attention_heads: + value: 4 +num_hidden_layers: + value: 94 +num_key_value_heads: + value: 4 +num_train_epochs: + value: 1 +optim: + value: adamw_torch_fused +optim_args: + value: null +optim_target_modules: + value: null +output_attentions: + value: false +output_dir: + value: out/glu-waleed-94L_run +output_hidden_states: + value: false +pad_token_id: + value: 0 +parallelism_config: + value: null +per_device_eval_batch_size: + value: 128 +per_device_train_batch_size: + value: 128 +prediction_loss_only: + value: false +pretraining_tp: + value: 1 +problem_type: + value: null +project: + value: huggingface +push_to_hub: + value: true +remove_unused_columns: + value: false +report_to: + value: + - wandb +restore_callback_states_from_checkpoint: + value: false +resume_from_checkpoint: + value: null +return_dict: + value: true +rms_norm_eps: + value: 1e-06 +rope_parameters: + value: + rope_theta: 10000 + rope_type: default +run_name: + value: LM-glu-waleed-94L-15.9M-20260809-055724 +save_on_each_node: + value: false +save_only_model: + value: false +save_steps: + value: 100 +save_strategy: + value: steps +save_total_limit: + value: null +seed: + value: 42 +skip_memory_metrics: + value: true +tf32: + value: null +tie_word_embeddings: + value: true +tokenizer_name: + value: w-ahmad/tiny-stories-tokenizer +torch_compile: + value: false +torch_compile_backend: + value: null +torch_compile_mode: + value: null +torch_empty_cache_steps: + value: null +trackio_bucket_id: + value: null +trackio_space_id: + value: null +trackio_static_space_id: + value: null +train_sampling_strategy: + value: random +transformers_version: + value: 5.15.0.dev0 +use_cache: + value: false +use_cpu: + value: false +use_liger_kernel: + value: false +vocab_size: + value: 4096 +warmup_steps: + value: 0 +weight_decay: + value: 0.01 diff --git a/wandb/run-20260809_055726-m1dnjnh6/files/output.log b/wandb/run-20260809_055726-m1dnjnh6/files/output.log new file mode 100644 index 0000000000000000000000000000000000000000..f4d966b802976b9dd6efcbaaa6957ef15140e3ed --- /dev/null +++ b/wandb/run-20260809_055726-m1dnjnh6/files/output.log @@ -0,0 +1,221 @@ +[transformers] `use_return_dict` is deprecated! Use `return_dict` instead! +[INFO] Causal mask (float with -inf) applied to all attention layers. +/mnt/data/zainulabideen/zain-exp/notebooks/Activation/exp.py:487: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1831.) + "std": tensor.std().item(), + 7%|██▌ | 100/1500 [05:06<57:27, 2.46s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '28.45', 'grad_norm': '3.625', 'learning_rate': '0.001', 'epoch': '0.01078', 'train/total_time_seconds': '49.92', 'train/time_per_step_avg': '2.496', 'train/epoch_time_elapsed': '58.55', 'train/estimated_remaining_minutes': '61.56', 'train/global/act/norm': '8.924e+04', 'train/global/act/mean': '-0.002389', 'train/global/act/std': '0.3927', 'train/global/act/max_abs': '8.323', 'train/global/act/frac_near_dtype_limit': '0', 'train/global/act/frac_near_user_limit': '0', 'train/global/grad/norm': '3.6', 'train/global/grad/mean': '-7.688e-08', 'train/global/grad/std': '0.0004508', 'train/global/grad/max_abs': '0.1279', 'train/global/grad/frac_near_dtype_limit': '0', 'train/global/grad/frac_near_user_limit': '0', 'train/global/param/norm': '174.8', 'train/global/param/mean': '0.001521', 'train/global/param/std': '0.04376', 'train/global/param/max_abs': '1', 'train/global/param/frac_near_dtype_limit': '0', 'train/global/param/frac_near_user_limit': '0', 'train/layer_model_layers_70/act/norm': '9250', 'train/layer_model_layers_70/act/mean': '-0.00157', 'train/layer_model_layers_70/act/std': '0.3992', 'train/layer_model_layers_70/act/max_abs': '4.719', 'train/layer_model_layers_70/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_70/act/frac_near_user_limit': '0', 'train/layer_model_layers_70/grad/norm': '0.1631', 'train/layer_model_layers_70/grad/mean': '1.986e-07', 'train/layer_model_layers_70/grad/std': '0.0002013', 'train/layer_model_layers_70/grad/max_abs': '0.006256', 'train/layer_model_layers_70/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_70/grad/frac_near_user_limit': '0', 'train/layer_model_layers_48/act/norm': '9114', 'train/layer_model_layers_48/act/mean': '-0.0005716', 'train/layer_model_layers_48/act/std': '0.3934', 'train/layer_model_layers_48/act/max_abs': '4.75', 'train/layer_model_layers_48/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_48/act/frac_near_user_limit': '0', 'train/layer_model_layers_48/grad/norm': '0.1751', 'train/layer_model_layers_48/grad/mean': '-3.435e-08', 'train/layer_model_layers_48/grad/std': '0.0002161', 'train/layer_model_layers_48/grad/max_abs': '0.005341', 'train/layer_model_layers_48/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_48/grad/frac_near_user_limit': '0', 'train/layer__model_layers_48/param/norm': '17.93', 'train/layer__model_layers_48/param/mean': '0.001577', 'train/layer__model_layers_48/param/std': '0.04424', 'train/layer__model_layers_48/param/max_abs': '1', 'train/layer__model_layers_48/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_48/param/frac_near_user_limit': '0', 'train/layer_model_layers_61/act/norm': '9197', 'train/layer_model_layers_61/act/mean': '7.977e-05', 'train/layer_model_layers_61/act/std': '0.397', 'train/layer_model_layers_61/act/max_abs': '4.688', 'train/layer_model_layers_61/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_61/act/frac_near_user_limit': '0', 'train/layer_model_layers_61/grad/norm': '0.1673', 'train/layer_model_layers_61/grad/mean': '-7.936e-08', 'train/layer_model_layers_61/grad/std': '0.0002065', 'train/layer_model_layers_61/grad/max_abs': '0.004791', 'train/layer_model_layers_61/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_61/grad/frac_near_user_limit': '0', 'train/layer__model_layers_88/param/norm': '17.94', 'train/layer__model_layers_88/param/mean': '0.001538', 'train/layer__model_layers_88/param/std': '0.04425', 'train/layer__model_layers_88/param/max_abs': '1', 'train/layer__model_layers_88/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_88/param/frac_near_user_limit': '0', 'train/layer_model_layers_23/act/norm': '9027', 'train/layer_model_layers_23/act/mean': '-0.006248', 'train/layer_model_layers_23/act/std': '0.3899', 'train/layer_model_layers_23/act/max_abs': '4.781', 'train/layer_model_layers_23/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_23/act/frac_near_user_limit': '0', 'train/layer_model_layers_23/grad/norm': '0.3035', 'train/layer_model_layers_23/grad/mean': '3.626e-08', 'train/layer_model_laye +{'loss': '24.26', 'grad_norm': '0.2314', 'learning_rate': '0.001', 'epoch': '0.02157', 'train/total_time_seconds': '93.08', 'train/time_per_step_avg': '2.327', 'train/epoch_time_elapsed': '109.9', 'train/estimated_remaining_minutes': '56.62'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '5.911', 'eval_runtime': '20.94', 'eval_samples_per_second': '455', 'eval_steps_per_second': '3.582', 'epoch': '0.02696', 'train/total_time_seconds': '115.2', 'train/time_per_step_avg': '2.304', 'train/epoch_time_elapsed': '156.9', 'train/estimated_remaining_minutes': '55.69'} +{'loss': '23.66', 'grad_norm': '8.75', 'learning_rate': '0.001', 'epoch': '0.03235', 'train/total_time_seconds': '136.5', 'train/time_per_step_avg': '2.275', 'train/epoch_time_elapsed': '182.2', 'train/estimated_remaining_minutes': '54.59'} +{'loss': '23.11', 'grad_norm': '4.156', 'learning_rate': '0.001', 'epoch': '0.04314', 'train/total_time_seconds': '180.3', 'train/time_per_step_avg': '2.254', 'train/epoch_time_elapsed': '233.9', 'train/estimated_remaining_minutes': '53.35'} +{'loss': '22.97', 'grad_norm': '1.578', 'learning_rate': '0.001', 'epoch': '0.05392', 'train/total_time_seconds': '224', 'train/time_per_step_avg': '2.24', 'train/epoch_time_elapsed': '285.5', 'train/estimated_remaining_minutes': '52.26'} +{'eval_loss': '5.726', 'eval_runtime': '20.4', 'eval_samples_per_second': '467.1', 'eval_steps_per_second': '3.677', 'epoch': '0.05392', 'train/total_time_seconds': '224', 'train/time_per_step_avg': '2.24', 'train/epoch_time_elapsed': '305.9', 'train/estimated_remaining_minutes': '52.26'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 14.32it/s] + 13%|█████▏ | 200/1500 [10:05<56:46, 2.62s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '22.91', 'grad_norm': '0.21', 'learning_rate': '0.001', 'epoch': '0.06471', 'train/total_time_seconds': '268.4', 'train/time_per_step_avg': '2.185', 'train/epoch_time_elapsed': '358.6', 'train/estimated_remaining_minutes': '51.45'} +{'loss': '22.9', 'grad_norm': '0.8594', 'learning_rate': '0.001', 'epoch': '0.07549', 'train/total_time_seconds': '311.5', 'train/time_per_step_avg': '2.184', 'train/epoch_time_elapsed': '409.9', 'train/estimated_remaining_minutes': '50.44'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '5.716', 'eval_runtime': '20.52', 'eval_samples_per_second': '464.3', 'eval_steps_per_second': '3.655', 'epoch': '0.08088', 'train/total_time_seconds': '333.7', 'train/time_per_step_avg': '2.185', 'train/epoch_time_elapsed': '456.6', 'train/estimated_remaining_minutes': '50.06'} +{'loss': '22.87', 'grad_norm': '5.312', 'learning_rate': '0.001', 'epoch': '0.08628', 'train/total_time_seconds': '355.8', 'train/time_per_step_avg': '2.193', 'train/epoch_time_elapsed': '482.5', 'train/estimated_remaining_minutes': '49.66'} +{'loss': '22.64', 'grad_norm': '3.516', 'learning_rate': '0.001', 'epoch': '0.09706', 'train/total_time_seconds': '398.8', 'train/time_per_step_avg': '2.185', 'train/epoch_time_elapsed': '533.2', 'train/estimated_remaining_minutes': '48.75'} +{'loss': '21.88', 'grad_norm': '2.797', 'learning_rate': '0.001', 'epoch': '0.1078', 'train/total_time_seconds': '443.1', 'train/time_per_step_avg': '2.192', 'train/epoch_time_elapsed': '585.3', 'train/estimated_remaining_minutes': '48.01'} +{'eval_loss': '5.379', 'eval_runtime': '19.95', 'eval_samples_per_second': '477.4', 'eval_steps_per_second': '3.759', 'epoch': '0.1078', 'train/total_time_seconds': '443.1', 'train/time_per_step_avg': '2.192', 'train/epoch_time_elapsed': '605.2', 'train/estimated_remaining_minutes': '48.01'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 13.21it/s] + 20%|███████▊ | 300/1500 [15:04<51:25, 2.57s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '21.07', 'grad_norm': '3', 'learning_rate': '0.001', 'epoch': '0.1186', 'train/total_time_seconds': '487.4', 'train/time_per_step_avg': '2.19', 'train/epoch_time_elapsed': '657.7', 'train/estimated_remaining_minutes': '47.26'} +{'loss': '20.38', 'grad_norm': '26.62', 'learning_rate': '0.001', 'epoch': '0.1294', 'train/total_time_seconds': '530.4', 'train/time_per_step_avg': '2.189', 'train/epoch_time_elapsed': '708.5', 'train/estimated_remaining_minutes': '46.41'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '4.97', 'eval_runtime': '20.85', 'eval_samples_per_second': '456.9', 'eval_steps_per_second': '3.597', 'epoch': '0.1348', 'train/total_time_seconds': '552.5', 'train/time_per_step_avg': '2.187', 'train/epoch_time_elapsed': '755.4', 'train/estimated_remaining_minutes': '46.04'} +{'loss': '19.99', 'grad_norm': '5.562', 'learning_rate': '0.001', 'epoch': '0.1402', 'train/total_time_seconds': '574.6', 'train/time_per_step_avg': '2.189', 'train/epoch_time_elapsed': '781.4', 'train/estimated_remaining_minutes': '45.68'} +{'loss': '19.31', 'grad_norm': '3.844', 'learning_rate': '0.001', 'epoch': '0.151', 'train/total_time_seconds': '617.5', 'train/time_per_step_avg': '2.187', 'train/epoch_time_elapsed': '832.1', 'train/estimated_remaining_minutes': '44.84'} +{'loss': '18.88', 'grad_norm': '8.188', 'learning_rate': '0.001', 'epoch': '0.1618', 'train/total_time_seconds': '661.7', 'train/time_per_step_avg': '2.186', 'train/epoch_time_elapsed': '883.9', 'train/estimated_remaining_minutes': '44.12'} +{'eval_loss': '4.667', 'eval_runtime': '20.01', 'eval_samples_per_second': '476.2', 'eval_steps_per_second': '3.749', 'epoch': '0.1618', 'train/total_time_seconds': '661.7', 'train/time_per_step_avg': '2.186', 'train/epoch_time_elapsed': '903.9', 'train/estimated_remaining_minutes': '44.12'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 18.19it/s] + 27%|██████████▍ | 400/1500 [20:04<47:52, 2.61s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '18.44', 'grad_norm': '8.688', 'learning_rate': '0.001', 'epoch': '0.1726', 'train/total_time_seconds': '706', 'train/time_per_step_avg': '2.186', 'train/epoch_time_elapsed': '956.6', 'train/estimated_remaining_minutes': '43.39'} +{'loss': '18.03', 'grad_norm': '8', 'learning_rate': '0.001', 'epoch': '0.1833', 'train/total_time_seconds': '749.5', 'train/time_per_step_avg': '2.191', 'train/epoch_time_elapsed': '1008', 'train/estimated_remaining_minutes': '42.62'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '4.424', 'eval_runtime': '20.62', 'eval_samples_per_second': '462.1', 'eval_steps_per_second': '3.638', 'epoch': '0.1887', 'train/total_time_seconds': '771.2', 'train/time_per_step_avg': '2.188', 'train/epoch_time_elapsed': '1054', 'train/estimated_remaining_minutes': '42.23'} +{'loss': '17.73', 'grad_norm': '9.938', 'learning_rate': '0.001', 'epoch': '0.1941', 'train/total_time_seconds': '793.6', 'train/time_per_step_avg': '2.189', 'train/epoch_time_elapsed': '1081', 'train/estimated_remaining_minutes': '41.88'} +{'loss': '17.36', 'grad_norm': '15.62', 'learning_rate': '0.001', 'epoch': '0.2049', 'train/total_time_seconds': '836.6', 'train/time_per_step_avg': '2.19', 'train/epoch_time_elapsed': '1132', 'train/estimated_remaining_minutes': '41.1'} +{'loss': '17.05', 'grad_norm': '4.719', 'learning_rate': '0.001', 'epoch': '0.2157', 'train/total_time_seconds': '880.7', 'train/time_per_step_avg': '2.189', 'train/epoch_time_elapsed': '1183', 'train/estimated_remaining_minutes': '40.36'} +{'eval_loss': '4.228', 'eval_runtime': '20.65', 'eval_samples_per_second': '461.3', 'eval_steps_per_second': '3.632', 'epoch': '0.2157', 'train/total_time_seconds': '880.7', 'train/time_per_step_avg': '2.189', 'train/epoch_time_elapsed': '1204', 'train/estimated_remaining_minutes': '40.36'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 17.91it/s] + 33%|█████████████ | 500/1500 [24:47<33:36, 2.02s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '16.77', 'grad_norm': '3.031', 'learning_rate': '0.001', 'epoch': '0.2265', 'train/total_time_seconds': '923.7', 'train/time_per_step_avg': '2.178', 'train/epoch_time_elapsed': '1255', 'train/estimated_remaining_minutes': '39.59'} +{'loss': '16.71', 'grad_norm': '16.62', 'learning_rate': '0.001', 'epoch': '0.2373', 'train/total_time_seconds': '966.9', 'train/time_per_step_avg': '2.175', 'train/epoch_time_elapsed': '1306', 'train/estimated_remaining_minutes': '38.82'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '4.116', 'eval_runtime': '17.5', 'eval_samples_per_second': '544.3', 'eval_steps_per_second': '4.285', 'epoch': '0.2427', 'train/total_time_seconds': '986.8', 'train/time_per_step_avg': '2.156', 'train/epoch_time_elapsed': '1348', 'train/estimated_remaining_minutes': '38.38'} +{'loss': '16.49', 'grad_norm': '9.625', 'learning_rate': '0.001', 'epoch': '0.248', 'train/total_time_seconds': '1008', 'train/time_per_step_avg': '2.144', 'train/epoch_time_elapsed': '1373', 'train/estimated_remaining_minutes': '37.98'} +{'loss': '16.2', 'grad_norm': '4.438', 'learning_rate': '0.001', 'epoch': '0.2588', 'train/total_time_seconds': '1053', 'train/time_per_step_avg': '2.16', 'train/epoch_time_elapsed': '1425', 'train/estimated_remaining_minutes': '37.28'} +{'loss': '15.98', 'grad_norm': '3.219', 'learning_rate': '0.001', 'epoch': '0.2696', 'train/total_time_seconds': '1089', 'train/time_per_step_avg': '2.085', 'train/epoch_time_elapsed': '1470', 'train/estimated_remaining_minutes': '36.31'} +{'eval_loss': '3.958', 'eval_runtime': '16.98', 'eval_samples_per_second': '561', 'eval_steps_per_second': '4.416', 'epoch': '0.2696', 'train/total_time_seconds': '1089', 'train/time_per_step_avg': '2.085', 'train/epoch_time_elapsed': '1487', 'train/estimated_remaining_minutes': '36.31'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 14.59it/s] +/mnt/data/zainulabideen/zain-exp/notebooks/Activation/exp.py:487: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1831.) + "std": tensor.std().item(), + 40%|███████████████▌ | 600/1500 [28:46<30:07, 2.01s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '15.73', 'grad_norm': '10.5', 'learning_rate': '0.001', 'epoch': '0.2804', 'train/total_time_seconds': '1124', 'train/time_per_step_avg': '2.007', 'train/epoch_time_elapsed': '1531', 'train/estimated_remaining_minutes': '35.32', 'train/global/act/norm': '5.349e+05', 'train/global/act/mean': '-0.02858', 'train/global/act/std': '2.354', 'train/global/act/max_abs': '78', 'train/global/act/frac_near_dtype_limit': '0', 'train/global/act/frac_near_user_limit': '9.816e-07', 'train/global/grad/norm': '2.817', 'train/global/grad/mean': '6.853e-08', 'train/global/grad/std': '0.0003526', 'train/global/grad/max_abs': '0.07227', 'train/global/grad/frac_near_dtype_limit': '0', 'train/global/grad/frac_near_user_limit': '0', 'train/global/param/norm': '192.8', 'train/global/param/mean': '0.001521', 'train/global/param/std': '0.04826', 'train/global/param/max_abs': '1', 'train/global/param/frac_near_dtype_limit': '0', 'train/global/param/frac_near_user_limit': '0', 'train/layer_model_layers_70/act/norm': '5.796e+04', 'train/layer_model_layers_70/act/mean': '0.01962', 'train/layer_model_layers_70/act/std': '2.5', 'train/layer_model_layers_70/act/max_abs': '44.25', 'train/layer_model_layers_70/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_70/act/frac_near_user_limit': '0', 'train/layer_model_layers_70/grad/norm': '0.02468', 'train/layer_model_layers_70/grad/mean': '-1.548e-07', 'train/layer_model_layers_70/grad/std': '3.047e-05', 'train/layer_model_layers_70/grad/max_abs': '0.0009537', 'train/layer_model_layers_70/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_70/grad/frac_near_user_limit': '0', 'train/layer_model_layers_48/act/norm': '5.606e+04', 'train/layer_model_layers_48/act/mean': '-0.006597', 'train/layer_model_layers_48/act/std': '2.419', 'train/layer_model_layers_48/act/max_abs': '44', 'train/layer_model_layers_48/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_48/act/frac_near_user_limit': '0', 'train/layer_model_layers_48/grad/norm': '0.04875', 'train/layer_model_layers_48/grad/mean': '1.199e-07', 'train/layer_model_layers_48/grad/std': '6.018e-05', 'train/layer_model_layers_48/grad/max_abs': '0.001808', 'train/layer_model_layers_48/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_48/grad/frac_near_user_limit': '0', 'train/layer__model_layers_48/param/norm': '19.09', 'train/layer__model_layers_48/param/mean': '0.00157', 'train/layer__model_layers_48/param/std': '0.0471', 'train/layer__model_layers_48/param/max_abs': '1', 'train/layer__model_layers_48/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_48/param/frac_near_user_limit': '0', 'train/layer_model_layers_61/act/norm': '5.571e+04', 'train/layer_model_layers_61/act/mean': '0.01297', 'train/layer_model_layers_61/act/std': '2.405', 'train/layer_model_layers_61/act/max_abs': '44', 'train/layer_model_layers_61/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_61/act/frac_near_user_limit': '0', 'train/layer_model_layers_61/grad/norm': '0.01859', 'train/layer_model_layers_61/grad/mean': '6.522e-08', 'train/layer_model_layers_61/grad/std': '2.294e-05', 'train/layer_model_layers_61/grad/max_abs': '0.0008545', 'train/layer_model_layers_61/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_61/grad/frac_near_user_limit': '0', 'train/layer__model_layers_88/param/norm': '20.96', 'train/layer__model_layers_88/param/mean': '0.001511', 'train/layer__model_layers_88/param/std': '0.05172', 'train/layer__model_layers_88/param/max_abs': '1', 'train/layer__model_layers_88/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_88/param/frac_near_user_limit': '0', 'train/layer_model_layers_23/act/norm': '4.959e+04', 'train/layer_model_layers_23/act/mean': '0.02837', 'train/layer_model_layers_23/act/std': '2.14', 'train/layer_model_layers_23/act/max_abs': '44.75', 'train/layer_model_layers_23/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_23/act/frac_near_user_limit': '0', 'train/layer_model_layers_23/grad/norm': '0.0559', 'train/layer_model_layers_23/grad/mean': '1.266e-08', 'train/layer_mode +{'loss': '15.58', 'grad_norm': '9.812', 'learning_rate': '0.001', 'epoch': '0.2912', 'train/total_time_seconds': '1157', 'train/time_per_step_avg': '1.9', 'train/epoch_time_elapsed': '1571', 'train/estimated_remaining_minutes': '34.28'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.862', 'eval_runtime': '16.97', 'eval_samples_per_second': '561.4', 'eval_steps_per_second': '4.419', 'epoch': '0.2966', 'train/total_time_seconds': '1173', 'train/time_per_step_avg': '1.864', 'train/epoch_time_elapsed': '1608', 'train/estimated_remaining_minutes': '33.77'} +{'loss': '15.38', 'grad_norm': '3.984', 'learning_rate': '0.001', 'epoch': '0.302', 'train/total_time_seconds': '1189', 'train/time_per_step_avg': '1.815', 'train/epoch_time_elapsed': '1628', 'train/estimated_remaining_minutes': '33.28'} +{'loss': '15.1', 'grad_norm': '2.75', 'learning_rate': '0.001', 'epoch': '0.3128', 'train/total_time_seconds': '1222', 'train/time_per_step_avg': '1.694', 'train/epoch_time_elapsed': '1668', 'train/estimated_remaining_minutes': '32.31'} +{'loss': '14.87', 'grad_norm': '5.75', 'learning_rate': '0.001', 'epoch': '0.3235', 'train/total_time_seconds': '1255', 'train/time_per_step_avg': '1.654', 'train/epoch_time_elapsed': '1708', 'train/estimated_remaining_minutes': '31.36'} +{'eval_loss': '3.684', 'eval_runtime': '17.14', 'eval_samples_per_second': '556', 'eval_steps_per_second': '4.377', 'epoch': '0.3235', 'train/total_time_seconds': '1255', 'train/time_per_step_avg': '1.654', 'train/epoch_time_elapsed': '1726', 'train/estimated_remaining_minutes': '31.36'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 19.57it/s] + 47%|██████████████████▏ | 700/1500 [32:41<26:46, 2.01s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '14.85', 'grad_norm': '5.281', 'learning_rate': '0.001', 'epoch': '0.3343', 'train/total_time_seconds': '1287', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '1766', 'train/estimated_remaining_minutes': '30.46'} +{'loss': '14.56', 'grad_norm': '5.125', 'learning_rate': '0.001', 'epoch': '0.3451', 'train/total_time_seconds': '1320', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '1807', 'train/estimated_remaining_minutes': '29.56'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.585', 'eval_runtime': '16.99', 'eval_samples_per_second': '560.9', 'eval_steps_per_second': '4.415', 'epoch': '0.3505', 'train/total_time_seconds': '1336', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '1844', 'train/estimated_remaining_minutes': '29.12'} +{'loss': '14.38', 'grad_norm': '7.375', 'learning_rate': '0.001', 'epoch': '0.3559', 'train/total_time_seconds': '1353', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '1864', 'train/estimated_remaining_minutes': '28.69'} +{'loss': '14.22', 'grad_norm': '4.062', 'learning_rate': '0.001', 'epoch': '0.3667', 'train/total_time_seconds': '1385', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '1904', 'train/estimated_remaining_minutes': '27.84'} +{'loss': '14.13', 'grad_norm': '5.062', 'learning_rate': '0.001', 'epoch': '0.3775', 'train/total_time_seconds': '1418', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '1944', 'train/estimated_remaining_minutes': '27'} +{'eval_loss': '3.524', 'eval_runtime': '16.92', 'eval_samples_per_second': '563.1', 'eval_steps_per_second': '4.433', 'epoch': '0.3775', 'train/total_time_seconds': '1418', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '1961', 'train/estimated_remaining_minutes': '27'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 15.10it/s] + 53%|████████████████████▊ | 800/1500 [36:38<23:23, 2.01s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '13.94', 'grad_norm': '6.594', 'learning_rate': '0.001', 'epoch': '0.3882', 'train/total_time_seconds': '1450', 'train/time_per_step_avg': '1.627', 'train/epoch_time_elapsed': '2002', 'train/estimated_remaining_minutes': '26.18'} +{'loss': '13.77', 'grad_norm': '4.062', 'learning_rate': '0.001', 'epoch': '0.399', 'train/total_time_seconds': '1483', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '2043', 'train/estimated_remaining_minutes': '25.39'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.42', 'eval_runtime': '17.11', 'eval_samples_per_second': '556.7', 'eval_steps_per_second': '4.383', 'epoch': '0.4044', 'train/total_time_seconds': '1499', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '2080', 'train/estimated_remaining_minutes': '24.99'} +{'loss': '13.64', 'grad_norm': '4.094', 'learning_rate': '0.001', 'epoch': '0.4098', 'train/total_time_seconds': '1516', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '2100', 'train/estimated_remaining_minutes': '24.6'} +{'loss': '13.44', 'grad_norm': '9.062', 'learning_rate': '0.001', 'epoch': '0.4206', 'train/total_time_seconds': '1548', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '2140', 'train/estimated_remaining_minutes': '23.82'} +{'loss': '13.21', 'grad_norm': '2.281', 'learning_rate': '0.001', 'epoch': '0.4314', 'train/total_time_seconds': '1581', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '2180', 'train/estimated_remaining_minutes': '23.05'} +{'eval_loss': '3.366', 'eval_runtime': '17.02', 'eval_samples_per_second': '559.9', 'eval_steps_per_second': '4.408', 'epoch': '0.4314', 'train/total_time_seconds': '1581', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '2197', 'train/estimated_remaining_minutes': '23.05'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 17.71it/s] + 60%|███████████████████████▍ | 900/1500 [40:35<20:08, 2.01s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '13.2', 'grad_norm': '5.125', 'learning_rate': '0.001', 'epoch': '0.4422', 'train/total_time_seconds': '1613', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '2239', 'train/estimated_remaining_minutes': '22.3'} +{'loss': '12.95', 'grad_norm': '2.672', 'learning_rate': '0.001', 'epoch': '0.453', 'train/total_time_seconds': '1646', 'train/time_per_step_avg': '1.627', 'train/epoch_time_elapsed': '2279', 'train/estimated_remaining_minutes': '21.55'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '3.194', 'eval_runtime': '17.01', 'eval_samples_per_second': '559.9', 'eval_steps_per_second': '4.408', 'epoch': '0.4583', 'train/total_time_seconds': '1662', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '2316', 'train/estimated_remaining_minutes': '21.19'} +{'loss': '12.77', 'grad_norm': '2.688', 'learning_rate': '0.001', 'epoch': '0.4637', 'train/total_time_seconds': '1679', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '2336', 'train/estimated_remaining_minutes': '20.82'} +{'loss': '12.55', 'grad_norm': '2.812', 'learning_rate': '0.001', 'epoch': '0.4745', 'train/total_time_seconds': '1711', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '2377', 'train/estimated_remaining_minutes': '20.09'} +{'loss': '12.37', 'grad_norm': '17.12', 'learning_rate': '0.001', 'epoch': '0.4853', 'train/total_time_seconds': '1744', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '2417', 'train/estimated_remaining_minutes': '19.37'} +{'eval_loss': '3.101', 'eval_runtime': '17.14', 'eval_samples_per_second': '556', 'eval_steps_per_second': '4.377', 'epoch': '0.4853', 'train/total_time_seconds': '1744', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '2434', 'train/estimated_remaining_minutes': '19.37'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 14.90it/s] + 67%|█████████████████████████▎ | 1000/1500 [44:31<16:45, 2.01s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '12.21', 'grad_norm': '3.141', 'learning_rate': '0.001', 'epoch': '0.4961', 'train/total_time_seconds': '1776', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '2475', 'train/estimated_remaining_minutes': '18.66'} +{'loss': '12.04', 'grad_norm': '3.344', 'learning_rate': '0.001', 'epoch': '0.5069', 'train/total_time_seconds': '1809', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '2515', 'train/estimated_remaining_minutes': '17.96'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.986', 'eval_runtime': '17.02', 'eval_samples_per_second': '559.8', 'eval_steps_per_second': '4.407', 'epoch': '0.5123', 'train/total_time_seconds': '1825', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '2552', 'train/estimated_remaining_minutes': '17.61'} +{'loss': '11.97', 'grad_norm': '4.125', 'learning_rate': '0.001', 'epoch': '0.5177', 'train/total_time_seconds': '1841', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '2573', 'train/estimated_remaining_minutes': '17.26'} +{'loss': '11.83', 'grad_norm': '6.062', 'learning_rate': '0.001', 'epoch': '0.5284', 'train/total_time_seconds': '1874', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '2613', 'train/estimated_remaining_minutes': '16.57'} +{'loss': '11.67', 'grad_norm': '2.281', 'learning_rate': '0.001', 'epoch': '0.5392', 'train/total_time_seconds': '1907', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '2653', 'train/estimated_remaining_minutes': '15.89'} +{'eval_loss': '2.914', 'eval_runtime': '17.02', 'eval_samples_per_second': '559.6', 'eval_steps_per_second': '4.406', 'epoch': '0.5392', 'train/total_time_seconds': '1907', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '2670', 'train/estimated_remaining_minutes': '15.89'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 19.04it/s] +/mnt/data/zainulabideen/zain-exp/notebooks/Activation/exp.py:487: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1831.) + "std": tensor.std().item(), + 73%|███████████████████████████▊ | 1100/1500 [48:29<13:24, 2.01s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '11.57', 'grad_norm': '3.719', 'learning_rate': '0.001', 'epoch': '0.55', 'train/total_time_seconds': '1942', 'train/time_per_step_avg': '1.657', 'train/epoch_time_elapsed': '2714', 'train/estimated_remaining_minutes': '15.23', 'train/global/act/norm': '2.376e+05', 'train/global/act/mean': '-0.03911', 'train/global/act/std': '1.045', 'train/global/act/max_abs': '46.5', 'train/global/act/frac_near_dtype_limit': '0', 'train/global/act/frac_near_user_limit': '0', 'train/global/grad/norm': '2.779', 'train/global/grad/mean': '-6.876e-08', 'train/global/grad/std': '0.0003479', 'train/global/grad/max_abs': '0.05688', 'train/global/grad/frac_near_dtype_limit': '0', 'train/global/grad/frac_near_user_limit': '0', 'train/global/param/norm': '213', 'train/global/param/mean': '0.001536', 'train/global/param/std': '0.05333', 'train/global/param/max_abs': '1', 'train/global/param/frac_near_dtype_limit': '0', 'train/global/param/frac_near_user_limit': '0', 'train/layer_model_layers_70/act/norm': '2.084e+04', 'train/layer_model_layers_70/act/mean': '0.01399', 'train/layer_model_layers_70/act/std': '0.899', 'train/layer_model_layers_70/act/max_abs': '25.25', 'train/layer_model_layers_70/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_70/act/frac_near_user_limit': '0', 'train/layer_model_layers_70/grad/norm': '0.05673', 'train/layer_model_layers_70/grad/mean': '-1.181e-08', 'train/layer_model_layers_70/grad/std': '7.004e-05', 'train/layer_model_layers_70/grad/max_abs': '0.001945', 'train/layer_model_layers_70/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_70/grad/frac_near_user_limit': '0', 'train/layer_model_layers_48/act/norm': '2.06e+04', 'train/layer_model_layers_48/act/mean': '0.0002608', 'train/layer_model_layers_48/act/std': '0.8887', 'train/layer_model_layers_48/act/max_abs': '24', 'train/layer_model_layers_48/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_48/act/frac_near_user_limit': '0', 'train/layer_model_layers_48/grad/norm': '0.05406', 'train/layer_model_layers_48/grad/mean': '2.672e-08', 'train/layer_model_layers_48/grad/std': '6.672e-05', 'train/layer_model_layers_48/grad/max_abs': '0.002243', 'train/layer_model_layers_48/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_48/grad/frac_near_user_limit': '0', 'train/layer__model_layers_48/param/norm': '20.01', 'train/layer__model_layers_48/param/mean': '0.001584', 'train/layer__model_layers_48/param/std': '0.04934', 'train/layer__model_layers_48/param/max_abs': '1', 'train/layer__model_layers_48/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_48/param/frac_near_user_limit': '0', 'train/layer_model_layers_61/act/norm': '1.996e+04', 'train/layer_model_layers_61/act/mean': '0.003131', 'train/layer_model_layers_61/act/std': '0.8611', 'train/layer_model_layers_61/act/max_abs': '24.5', 'train/layer_model_layers_61/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_61/act/frac_near_user_limit': '0', 'train/layer_model_layers_61/grad/norm': '0.04217', 'train/layer_model_layers_61/grad/mean': '-9.741e-08', 'train/layer_model_layers_61/grad/std': '5.206e-05', 'train/layer_model_layers_61/grad/max_abs': '0.001236', 'train/layer_model_layers_61/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_61/grad/frac_near_user_limit': '0', 'train/layer__model_layers_88/param/norm': '25.49', 'train/layer__model_layers_88/param/mean': '0.001532', 'train/layer__model_layers_88/param/std': '0.06291', 'train/layer__model_layers_88/param/max_abs': '1', 'train/layer__model_layers_88/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_88/param/frac_near_user_limit': '0', 'train/layer_model_layers_23/act/norm': '2.166e+04', 'train/layer_model_layers_23/act/mean': '0.007417', 'train/layer_model_layers_23/act/std': '0.9344', 'train/layer_model_layers_23/act/max_abs': '27', 'train/layer_model_layers_23/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_23/act/frac_near_user_limit': '0', 'train/layer_model_layers_23/grad/norm': '0.07506', 'train/layer_model_layers_23/grad/mean': '1.411e-07', 'train/layer_mode +{'loss': '11.48', 'grad_norm': '4.281', 'learning_rate': '0.001', 'epoch': '0.5608', 'train/total_time_seconds': '1974', 'train/time_per_step_avg': '1.657', 'train/epoch_time_elapsed': '2754', 'train/estimated_remaining_minutes': '14.56'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.829', 'eval_runtime': '16.92', 'eval_samples_per_second': '563.2', 'eval_steps_per_second': '4.434', 'epoch': '0.5662', 'train/total_time_seconds': '1991', 'train/time_per_step_avg': '1.657', 'train/epoch_time_elapsed': '2791', 'train/estimated_remaining_minutes': '14.22'} +{'loss': '11.31', 'grad_norm': '3.328', 'learning_rate': '0.001', 'epoch': '0.5716', 'train/total_time_seconds': '2007', 'train/time_per_step_avg': '1.657', 'train/epoch_time_elapsed': '2812', 'train/estimated_remaining_minutes': '13.89'} +{'loss': '11.14', 'grad_norm': '2.906', 'learning_rate': '0.001', 'epoch': '0.5824', 'train/total_time_seconds': '2040', 'train/time_per_step_avg': '1.655', 'train/epoch_time_elapsed': '2852', 'train/estimated_remaining_minutes': '13.22'} +{'loss': '10.99', 'grad_norm': '2.438', 'learning_rate': '0.001', 'epoch': '0.5932', 'train/total_time_seconds': '2072', 'train/time_per_step_avg': '1.655', 'train/epoch_time_elapsed': '2892', 'train/estimated_remaining_minutes': '12.56'} +{'eval_loss': '2.726', 'eval_runtime': '17.1', 'eval_samples_per_second': '557.2', 'eval_steps_per_second': '4.387', 'epoch': '0.5932', 'train/total_time_seconds': '2072', 'train/time_per_step_avg': '1.655', 'train/epoch_time_elapsed': '2909', 'train/estimated_remaining_minutes': '12.56'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 13.76it/s] + 80%|██████████████████████████████▍ | 1200/1500 [52:25<10:00, 2.00s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '10.87', 'grad_norm': '2.953', 'learning_rate': '0.001', 'epoch': '0.6039', 'train/total_time_seconds': '2105', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '2950', 'train/estimated_remaining_minutes': '11.9'} +{'loss': '10.78', 'grad_norm': '3.172', 'learning_rate': '0.001', 'epoch': '0.6147', 'train/total_time_seconds': '2137', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '2990', 'train/estimated_remaining_minutes': '11.25'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.652', 'eval_runtime': '17', 'eval_samples_per_second': '560.5', 'eval_steps_per_second': '4.413', 'epoch': '0.6201', 'train/total_time_seconds': '2154', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '3027', 'train/estimated_remaining_minutes': '10.92'} +{'loss': '10.61', 'grad_norm': '2.391', 'learning_rate': '0.001', 'epoch': '0.6255', 'train/total_time_seconds': '2170', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '3048', 'train/estimated_remaining_minutes': '10.6'} +{'loss': '10.52', 'grad_norm': '2.688', 'learning_rate': '0.001', 'epoch': '0.6363', 'train/total_time_seconds': '2202', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '3088', 'train/estimated_remaining_minutes': '9.954'} +{'loss': '10.34', 'grad_norm': '2.391', 'learning_rate': '0.001', 'epoch': '0.6471', 'train/total_time_seconds': '2235', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '3128', 'train/estimated_remaining_minutes': '9.312'} +{'eval_loss': '2.583', 'eval_runtime': '16.99', 'eval_samples_per_second': '560.7', 'eval_steps_per_second': '4.414', 'epoch': '0.6471', 'train/total_time_seconds': '2235', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '3145', 'train/estimated_remaining_minutes': '9.312'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 18.75it/s] + 87%|████████████████████████████████▉ | 1300/1500 [56:21<06:43, 2.02s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '10.24', 'grad_norm': '2.688', 'learning_rate': '0.001', 'epoch': '0.6579', 'train/total_time_seconds': '2267', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '3186', 'train/estimated_remaining_minutes': '8.673'} +{'loss': '10.12', 'grad_norm': '2.141', 'learning_rate': '0.001', 'epoch': '0.6686', 'train/total_time_seconds': '2300', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '3226', 'train/estimated_remaining_minutes': '8.038'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.531', 'eval_runtime': '16.9', 'eval_samples_per_second': '563.7', 'eval_steps_per_second': '4.438', 'epoch': '0.674', 'train/total_time_seconds': '2316', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '3263', 'train/estimated_remaining_minutes': '7.721'} +{'loss': '10.03', 'grad_norm': '4.156', 'learning_rate': '0.001', 'epoch': '0.6794', 'train/total_time_seconds': '2333', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '3284', 'train/estimated_remaining_minutes': '7.406'} +{'loss': '9.931', 'grad_norm': '2.781', 'learning_rate': '0.001', 'epoch': '0.6902', 'train/total_time_seconds': '2365', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '3324', 'train/estimated_remaining_minutes': '6.776'} +{'loss': '9.83', 'grad_norm': '3.578', 'learning_rate': '0.001', 'epoch': '0.701', 'train/total_time_seconds': '2398', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '3364', 'train/estimated_remaining_minutes': '6.149'} +{'eval_loss': '2.456', 'eval_runtime': '16.95', 'eval_samples_per_second': '562', 'eval_steps_per_second': '4.424', 'epoch': '0.701', 'train/total_time_seconds': '2398', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '3381', 'train/estimated_remaining_minutes': '6.149'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 14.16it/s] + 93%|█████████████████████████████████▌ | 1400/1500 [1:00:17<03:22, 2.02s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '9.744', 'grad_norm': '2.062', 'learning_rate': '0.001', 'epoch': '0.7118', 'train/total_time_seconds': '2431', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '3422', 'train/estimated_remaining_minutes': '5.524'} +{'loss': '9.647', 'grad_norm': '4.719', 'learning_rate': '0.001', 'epoch': '0.7226', 'train/total_time_seconds': '2463', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '3462', 'train/estimated_remaining_minutes': '4.902'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.387', 'eval_runtime': '17.14', 'eval_samples_per_second': '555.7', 'eval_steps_per_second': '4.375', 'epoch': '0.728', 'train/total_time_seconds': '2479', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '3499', 'train/estimated_remaining_minutes': '4.591'} +{'loss': '9.56', 'grad_norm': '2.797', 'learning_rate': '0.001', 'epoch': '0.7334', 'train/total_time_seconds': '2496', 'train/time_per_step_avg': '1.627', 'train/epoch_time_elapsed': '3519', 'train/estimated_remaining_minutes': '4.282'} +{'loss': '9.471', 'grad_norm': '1.969', 'learning_rate': '0.001', 'epoch': '0.7441', 'train/total_time_seconds': '2528', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '3559', 'train/estimated_remaining_minutes': '3.664'} +{'loss': '9.409', 'grad_norm': '2.906', 'learning_rate': '0.001', 'epoch': '0.7549', 'train/total_time_seconds': '2561', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '3600', 'train/estimated_remaining_minutes': '3.049'} +{'eval_loss': '2.356', 'eval_runtime': '17.24', 'eval_samples_per_second': '552.7', 'eval_steps_per_second': '4.351', 'epoch': '0.7549', 'train/total_time_seconds': '2561', 'train/time_per_step_avg': '1.63', 'train/epoch_time_elapsed': '3617', 'train/estimated_remaining_minutes': '3.049'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 18.82it/s] +100%|████████████████████████████████████| 1500/1500 [1:04:14<00:00, 2.01s/it][transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. +{'loss': '9.326', 'grad_norm': '2.531', 'learning_rate': '0.001', 'epoch': '0.7657', 'train/total_time_seconds': '2594', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '3658', 'train/estimated_remaining_minutes': '2.435'} +{'loss': '9.252', 'grad_norm': '2.312', 'learning_rate': '0.001', 'epoch': '0.7765', 'train/total_time_seconds': '2626', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '3698', 'train/estimated_remaining_minutes': '1.824'} + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes +{'eval_loss': '2.293', 'eval_runtime': '17.2', 'eval_samples_per_second': '554', 'eval_steps_per_second': '4.361', 'epoch': '0.7819', 'train/total_time_seconds': '2642', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '3736', 'train/estimated_remaining_minutes': '1.519'} +{'loss': '9.142', 'grad_norm': '1.984', 'learning_rate': '0.001', 'epoch': '0.7873', 'train/total_time_seconds': '2659', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '3756', 'train/estimated_remaining_minutes': '1.214'} +{'loss': '9.064', 'grad_norm': '1.656', 'learning_rate': '0.001', 'epoch': '0.7981', 'train/total_time_seconds': '2691', 'train/time_per_step_avg': '1.631', 'train/epoch_time_elapsed': '3796', 'train/estimated_remaining_minutes': '0.6061'} +{'loss': '8.985', 'grad_norm': '1.727', 'learning_rate': '0.001', 'epoch': '0.8088', 'train/total_time_seconds': '2724', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '3837', 'train/estimated_remaining_minutes': '0'} +{'eval_loss': '2.24', 'eval_runtime': '17.03', 'eval_samples_per_second': '559.3', 'eval_steps_per_second': '4.403', 'epoch': '0.8088', 'train/total_time_seconds': '2724', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '3854', 'train/estimated_remaining_minutes': '0'} + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 14.29it/s] +100%|████████████████████████████████████| 1500/1500 [1:04:14<00:00, 2.57s/it] +{'train_runtime': '3855', 'train_samples_per_second': '199.2', 'train_steps_per_second': '0.389', 'train_loss': '14.66', 'epoch': '0.8088', 'train/total_time_seconds': '2724', 'train/time_per_step_avg': '1.628', 'train/epoch_time_elapsed': '3854', 'train/estimated_remaining_minutes': '0'} +100%|██████████████████████████████████████████| 75/75 [00:16<00:00, 4.47it/s] +[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 16.52it/s] +[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 16.30it/s] +Found 7 files to upload + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████░░░░░░░░░░░░ 3 / 7 + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████░░░░░░░░░░░░ 3 / 7 + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████████████████ 7 / 7 ✓ +[transformers] TinyLlamaForCausalLM has generative capabilities, as `prepare_inputs_for_generation` is explicitly defined. However, it doesn't directly inherit from `GenerationMixin`. From 👉v4.50👈 onwards, `PreTrainedModel` will NOT inherit from `GenerationMixin`, and this model will lose the ability to call `generate` and other related functions. + - If you're using `trust_remote_code=True`, you can get rid of this warning by loading the model with an auto class. See https://huggingface.co/docs/transformers/en/model_doc/auto#auto-classes + - If you are the owner of the model architecture code, please modify your model class such that it inherits from `GenerationMixin` (after `PreTrainedModel`, otherwise you'll get an exception). + - If you are not the owner of the model architecture class, please contact the model code owner to update it. +Writing model shards: 100%|██████████████████████| 1/1 [00:00<00:00, 2.85it/s] +Found 7 files to upload + Preparing ████████████████████ 7 / 7 ✓ + Uploading ████████████████████ 2 / 2 files ✓ + Committing ████████████████████ 7 / 7 ✓ +No files have been modified since last commit. Skipping to prevent empty commit. diff --git a/wandb/run-20260809_055726-m1dnjnh6/files/requirements.txt b/wandb/run-20260809_055726-m1dnjnh6/files/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..123b15ebf857859624f7f4332e92341c8ef13fdf --- /dev/null +++ b/wandb/run-20260809_055726-m1dnjnh6/files/requirements.txt @@ -0,0 +1,149 @@ +asttokens==3.0.1 +comm==0.2.3 +debugpy==1.8.21 +decorator==5.3.1 +executing==2.2.1 +nest-asyncio==1.6.0 +parso==0.8.7 +platformdirs==4.11.0 +psutil==7.2.2 +ptyprocess==0.7.0 +pure_eval==0.2.3 +Pygments==2.20.0 +pyzmq==27.1.0 +setuptools==83.0.0 +six==1.17.0 +tornado==6.5.7 +traitlets==5.15.0 +fsspec==2026.4.0 +wcwidth==0.8.2 +ipython_pygments_lexers==1.1.1 +jedi==0.20.0 +jupyter_core==5.9.1 +matplotlib-inline==0.2.2 +pexpect==4.9.0 +prompt_toolkit==3.0.53 +python-dateutil==2.9.0.post0 +stack_data==0.6.3 +wheel==0.47.0 +jupyter_client==8.9.1 +pip==26.1.2 +ipython==9.15.0 +ipykernel==7.2.0 +threadpoolctl==3.6.0 +pyparsing==3.3.2 +typing_extensions==4.15.0 +Jinja2==3.1.6 +narwhals==2.24.0 +kiwisolver==1.5.0 +joblib==1.5.3 +fonttools==4.63.0 +cycler==0.12.1 +scipy==1.17.1 +pandas==3.0.5 +contourpy==1.3.3 +scikit-learn==1.9.0 +matplotlib==3.11.1 +urllib3==2.7.0 +tqdm==4.70.0 +idna==3.18 +charset-normalizer==3.4.9 +certifi==2026.7.22 +requests==2.34.2 +seaborn==0.13.2 +uv==0.12.0 +shellingham==1.5.4 +mpmath==1.3.0 +attrs==26.1.0 +hf-xet==1.5.2 +nvidia-nccl-cu12==2.21.5 +MarkupSafe==3.0.3 +regex==2026.7.19 +importlib_metadata==9.0.0 +httpcore==1.0.9 +annotated-doc==0.0.5 +multidict==6.7.1 +aiohttp==3.14.3 +aiosignal==1.4.0 +xxhash==3.8.1 +aiohappyeyeballs==2.7.1 +mdurl==0.1.2 +cuda-toolkit==13.0.3.0 +networkx==3.6.1 +PyYAML==6.0.3 +nvidia-cufile==1.15.1.6 +typer==0.27.0 +torchaudio==2.6.0+cu124 +rich==15.0.0 +nvidia-cufft-cu12==11.2.1.3 +h11==0.16.0 +dill==0.4.1 +cuda-pathfinder==1.6.0 +filelock==3.29.0 +nvidia-nvtx-cu12==12.4.127 +httpx==0.28.1 +anyio==4.14.2 +numpy==2.4.4 +yarl==1.24.5 +click==8.4.2 +triton==3.2.0 +frozenlist==1.8.0 +zipp==4.1.0 +propcache==0.5.2 +tokenizers==0.22.2 +markdown-it-py==4.2.0 +nvidia-cuda-runtime==13.0.96 +cuda-bindings==13.3.1 +nvidia-cuda-cupti==13.0.85 +torch==2.6.0+cu124 +multiprocess==0.70.19 +pillow==12.2.0 +transformers==5.15.0.dev0 +wandb==0.28.1 +nvidia-curand==10.4.0.35 +sympy==1.13.1 +nvidia-cusparse==12.6.3.3 +nvidia-cuda-nvrtc==13.0.88 +typing-inspection==0.4.2 +nvidia-cusolver==12.0.4.66 +nvidia-cufft==12.0.0.61 +nvidia-cudnn-cu13==9.20.0.48 +nvidia-cublas==13.1.1.3 +pyarrow==25.0.0 +evaluate==0.4.6 +diffusers==0.39.0 +pydantic==2.13.4 +annotated-types==0.8.0 +protobuf==7.35.1 +sentry-sdk==2.66.1 +einops==0.8.2 +packaging==26.2 +nvidia-nvjitlink-cu12==12.4.127 +nvidia-curand-cu12==10.3.5.147 +nvidia-cusparselt-cu12==0.6.2 +nvidia-cusparse-cu12==12.3.1.170 +nvidia-cuda-runtime-cu12==12.4.127 +torchvision==0.21.0+cu124 +nvidia-cuda-nvrtc-cu12==12.4.127 +nvidia-cuda-cupti-cu12==12.4.127 +nvidia-cusolver-cu12==11.6.1.9 +nvidia-cublas-cu12==12.4.5.8 +nvidia-cudnn-cu12==9.1.0.70 +huggingface_hub==1.26.0 +datasets==5.0.1 +safetensors==0.8.0 +accelerate==1.14.0 +pydantic_core==2.46.4 +ninja==1.13.0 +autocommand==2.2.2 +backports.tarfile==1.2.0 +importlib_metadata==8.7.1 +jaraco.text==4.0.0 +jaraco.context==6.1.0 +jaraco.functools==4.4.0 +more-itertools==10.8.0 +packaging==26.0 +platformdirs==4.4.0 +tomli==2.4.0 +wheel==0.46.3 +zipp==3.23.0 diff --git a/wandb/run-20260809_055726-m1dnjnh6/files/wandb-metadata.json b/wandb/run-20260809_055726-m1dnjnh6/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..9e353150ecbeb2fe899c04bc4555c25486d44ebd --- /dev/null +++ b/wandb/run-20260809_055726-m1dnjnh6/files/wandb-metadata.json @@ -0,0 +1,96 @@ +{ + "os": "Linux-5.15.0-126-generic-x86_64-with-glibc2.35", + "python": "CPython 3.11.15", + "startedAt": "2026-08-09T05:57:26.362395Z", + "args": [ + "--config", + "configs/baseline.yaml", + "--variants", + "glu-waleed-94L", + "--push" + ], + "program": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py", + "codePath": "sweep.py", + "codePathLocal": "sweep.py", + "git": { + "remote": "https://github.com/deepnevro/Activation.git", + "commit": "34b8d2e8f9a0c5751333310e69fa0c1056381deb" + }, + "email": "deepnevro@gmail.com", + "root": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation", + "host": "deeplens-k3s-node1", + "executable": "/mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python", + "cpu_count": 112, + "cpu_count_logical": 224, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1560765693952", + "used": "708276572160" + } + }, + "memory": { + "total": "2164089937920" + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-39c684a5-fde6-83d7-1663-0859795881ae" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-132944c4-b689-2b5f-89a4-d730401677ab" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bc6c3e3c-9b90-09ca-c034-774961847c54" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea" + } + ], + "cudaVersion": "12.4", + "writerId": "k97nc6zhnp2eno8n59e5ngav280824l8" +} \ No newline at end of file diff --git a/wandb/run-20260809_055726-m1dnjnh6/files/wandb-summary.json b/wandb/run-20260809_055726-m1dnjnh6/files/wandb-summary.json new file mode 100644 index 0000000000000000000000000000000000000000..5d3a5a48c8387a74f96cf8d34f802515ddda1b24 --- /dev/null +++ b/wandb/run-20260809_055726-m1dnjnh6/files/wandb-summary.json @@ -0,0 +1 @@ +{"train/train/layer_model_layers_14/grad/mean":-6.321442782785889e-08,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/mean":-0.0001850128173828125,"train/train/tensor_act_model_layers_42_mlp_waleed_W_u/max_abs":2.390625,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/max_abs":0.00115966796875,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/mean":7.434027793351561e-08,"train/train/tensor_act_model_layers_14_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_mlp/max_abs":4.1875,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_mlp_waleed_W_g_weight/std":5.534079650330436e-05,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/mean":7.82012939453125e-05,"train/train/tensor_grad_model_layers_80_mlp_waleed_W_g_weight/std":7.678350291326782e-05,"train/train/layer_model_layers_15/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86/max_abs":31.875,"train/train/tensor_act_model_layers_55_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/std":0.037109375,"train/train/tensor_grad_model_layers_70_mlp_waleed_W_u_weight/max_abs":0.000514984130859375,"train/train/layer_model_layers_63/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_44_self_attn_o_proj/norm":328.80636347516275,"train/train/tensor_act_model_layers_82_self_attn_q_proj/mean":0.226806640625,"train/train/tensor_act_model_layers_29_post_attention_layernorm/mean":0.00012922286987304688,"train/train/tensor_act_model_layers_68_mlp_waleed_W_g/norm":3383.756287618519,"train/train/tensor_act_model_layers_74_self_attn_o_proj/max_abs":3.96875,"train/train/tensor_act_model_layers_37_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_77/grad/std":7.644767596841866e-05,"train/train/tensor_act_model_layers_73_mlp_waleed/mean":-0.002437591552734375,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/std":2.4846334282857304e-05,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/norm":4.125,"train/train/tensor_act_model_layers_7_self_attn/mean":-0.0009126663208007812,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/mean":-7.778406143188477e-06,"train/train/tensor_act_model_layers_54_mlp/mean":0.0002770423889160156,"train/train/tensor_param_model_layers_41_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_25/max_abs":26.5,"train/train/tensor_act_model_layers_52_mlp_down_proj/mean":0.0006361007690429688,"train/train/layer_model_layers_55/grad/norm":0.03588231490237581,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/norm":0.0010936611322505767,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/mean":7.175840437412262e-07,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/max_abs":0.0003509521484375,"train/train/tensor_param_model_layers_33_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_self_attn_q_proj/norm":5705.54486868424,"train/train/tensor_param_model_layers_40_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_85_mlp_waleed_W_g_weight/norm":8,"train/train/layer_model_layers_26/act/max_abs":26.375,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/mean":-6.079673767089844e-06,"train/train/tensor_act_model_layers_79_mlp_waleed_W_u/mean":-0.0019969940185546875,"train/train/tensor_grad_model_layers_88_mlp_waleed_W_u_weight/std":8.584593888198526e-05,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/std":0.00010458738140450106,"train/train/tensor_act_model_layers_55_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/max_abs":0.2578125,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/max_abs":0.0001850128173828125,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_mlp_down_proj/std":0.09741486417275942,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/max_abs":0.001007080078125,"train/train/tensor_act_model_layers_13_self_attn_v_proj/max_abs":1.828125,"train/train/tensor_act_model_layers_35_self_attn_q_proj/std":1.0781251848607665,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/max_abs":0.001007080078125,"train/train/tensor_act_model_layers_12_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/max_abs":0.000827789306640625,"train/train/layer_model_layers_93/grad/std":0.00011634522842683167,"train/train/tensor_act_model_layers_71_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/mean":2.1740561351180077e-08,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/max_abs":0.0003662109375,"train/train/layer_model_layers_79/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/std":4.8997806514605153e-05,"train/train/tensor_act_model_layers_52_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_mlp_waleed_W_u/std":0.26367198515827167,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/std":0.041259765625,"train/train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_mlp_waleed_W_g/norm":3716.6741199365188,"train/train/tensor_act_model_layers_14_mlp_waleed_W_u/max_abs":3.078125,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_12_mlp_waleed_W_u/std":0.25293158295862755,"train/train/tensor_act_model_layers_52_input_layernorm/max_abs":5.03125,"train/train/tensor_act_model_layers_72_self_attn_k_proj/max_abs":6.875,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/norm":0.0009129507843245149,"train/train/tensor_act_model_layers_14_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/mean":0.0004253387451171875,"train/train/tensor_act_model_layers_16_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_mlp_waleed_W_u_weight/mean":-5.299225449562073e-07,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/max_abs":0.00090789794921875,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp_waleed_W_g/mean":-0.002410888671875,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_v_proj/norm":2134.409557787317,"train/train/layer_model_layers_2/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp_down_proj/max_abs":0.46484375,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/norm":3.703125,"train/train/tensor_param_model_layers_86_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/norm":0.0007508512340903508,"train/train/tensor_param_model_layers_54_mlp_waleed_W_g_weight/mean":2.3365020751953125e-05,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/mean":-3.0040740966796875e-05,"train/train/layer__model_layers_53/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/mean":5.8673322200775146e-08,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/mean":0.0003299713134765625,"train/train/tensor_act_model_layers_63_self_attn_k_proj/max_abs":5.84375,"train/train/tensor_act_model_layers_31_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn_k_proj/max_abs":4.34375,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/norm":0.006256587312181497,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/mean":4.057073965668678e-08,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp_down_proj/mean":0.00223541259765625,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/mean":0.00061798095703125,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/mean":0.00024127960205078125,"train/train/tensor_param_model_layers_89_mlp_waleed_W_g_weight/std":0.05029296875,"train/train/tensor_act_model_layers_86_self_attn_o_proj/max_abs":5.90625,"train/train/tensor_act_model_layers_90_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp/norm":1777.9786401867384,"train/train/tensor_param_model_layers_57_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/norm":4.90625,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/max_abs":0.09228515625,"train/train/tensor_act_model_layers_46_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_78_mlp_waleed/norm":2012.6290024078883,"train/train/tensor_act_model_layers_26_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_post_attention_layernorm/norm":5792.607543947518,"train/train/tensor_act_model_layers_78_mlp_waleed_W_u/max_abs":3.5,"train/train/tensor_param_model_layers_48_mlp_waleed_W_u_weight/mean":0.0001354217529296875,"train/train/tensor_param_model_layers_7_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_62_mlp_waleed_W_g_weight/mean":0.0001926422119140625,"train/train/tensor_act_model_layers_83_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_45_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_73_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/mean":-0.00019931793212890625,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/mean":-3.4905970096588135e-06,"train/train/tensor_act_model_layers_81_post_attention_layernorm/max_abs":5.15625,"train/train/tensor_param_model_layers_38_mlp_waleed_W_u_weight/max_abs":0.1337890625,"train/train/layer_model_layers_4/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_65/act/std":0.8597946325129263,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_mlp_down_proj/max_abs":0.96484375,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/norm":0.014467070269869311,"train/train/tensor_act_model_layers_28_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_o_proj/mean":-0.0009126663208007812,"train/train/tensor_grad_model_layers_22_mlp_waleed_W_g_weight/max_abs":0.001434326171875,"train/train/layer_model_layers_33/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_mlp_waleed_W_u_weight/max_abs":0.11279296875,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/mean":0.00019741058349609375,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn_o_proj/max_abs":1.375,"train/train/tensor_act_model_layers_84_mlp_waleed/mean":-0.00333404541015625,"train/train/tensor_act_model_layers_56_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn_o_proj/max_abs":3.65625,"train/train/tensor_grad_model_layers_77_mlp_waleed_W_g_weight/norm":0.02334808140393611,"train/train/tensor_act_model_layers_83_input_layernorm/max_abs":5.375,"train/train/tensor_act_model_layers_92_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/std":0.039794921875,"train/train/tensor_grad_model_layers_39_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/norm":5.6875,"train/train/tensor_grad_model_layers_90_mlp_waleed_W_g_weight/norm":0.03730701219595255,"train/train/layer_model_layers_71/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30/max_abs":26.375,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/norm":0.01840480702286344,"train/train/tensor_grad_model_layers_21_mlp_waleed_W_u_weight/std":5.14477958923476e-05,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_o_proj/norm":1836.6584070356207,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/mean":7.963180541992188e-05,"train/train/tensor_act_model_layers_6_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_mlp_waleed_W_g_weight/mean":0.000125885009765625,"train/train/tensor_act_model_layers_48_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_mlp_waleed_W_g_weight/mean":2.591405063867569e-07,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/norm":6.875,"train/train/tensor_act_model_layers_59_self_attn_q_proj/std":1.056646075560778,"train/train/tensor_act_model_layers_6_mlp/std":0.3027438357436181,"train/train/tensor_act_model_layers_16_self_attn_o_proj/std":0.07617740194094678,"train/train/tensor_act_model_layers_11_self_attn_q_proj/std":1.4062500417232506,"train/train/tensor_act_model_layers_87_self_attn/mean":-0.008819580078125,"train/train/layer_model_layers_53/act/mean":0.005659548100084066,"train/train/tensor_act_model_layers_87_mlp_waleed/mean":0.008514404296875,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/std":7.86597061563359e-05,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_q_proj/mean":0.0989990234375,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_6/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp_down_proj/mean":-0.0014400482177734375,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/mean":-1.2731179594993591e-06,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/max_abs":0.298828125,"train/train/layer_model_layers_1/grad/mean":-2.68284830027921e-07,"train/train/tensor_act_model_layers_32_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24/max_abs":26.5,"train/train/tensor_grad_model_layers_45_mlp_waleed_W_u_weight/std":3.820792959581796e-05,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/max_abs":0.000579833984375,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/norm":0.03339443071211672,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/norm":0.0006555371618986176,"train/train/tensor_param_model_layers_16_mlp_waleed_W_u_weight/mean":7.152557373046875e-06,"train/train/tensor_param_model_layers_26_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/norm":4.65625,"train/train/tensor_act_model_layers_52_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/max_abs":0.1962890625,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/max_abs":0.12353515625,"train/train/tensor_act_model_layers_26/std":2.742243611457142,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/std":0.025634765625,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/norm":3.09375,"train/train/tensor_act_model_layers_75_mlp_waleed/mean":0.0017337799072265625,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/mean":1.484295353293419e-07,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/max_abs":0.2890625,"train/train/tensor_act_model_layers_0_self_attn_o_proj/mean":0.0132904052734375,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/max_abs":0.2333984375,"train/train/tensor_act_model_layers_30_mlp_waleed_W_g/max_abs":1.9375,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/max_abs":0.0003566741943359375,"train/train/tensor_act_model_layers_14_self_attn_q_proj/norm":5510.765452622139,"train/train/layer__model_layers_10/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/std":0.035888671875,"train/train/layer_model_layers_55/act/mean":0.0015951432287693024,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/std":5.9251074883717905e-05,"train/train/tensor_act_model_layers_44_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_mlp_down_proj/max_abs":24.5,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/max_abs":0.2578125,"train/train/tensor_param_model_layers_39_mlp_waleed_W_g_weight/std":0.0260009765625,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/norm":4.90625,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/max_abs":0.00115966796875,"train/train/layer_model_layers_92/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_input_layernorm/max_abs":5.15625,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/max_abs":0.00013637542724609375,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/std":4.005678256596027e-05,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_41/param/std":0.0481406114844343,"train/train/tensor_act_model_layers_6_mlp_down_proj/std":0.3027438357436181,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/mean":-0.0001074075698852539,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_mlp_waleed_W_g/mean":0.015625,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/std":0.0361328125,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/norm":3.359375,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/norm":4.59375,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/mean":4.393979907035828e-06,"train/train/layer_model_layers_14/grad/norm":0.04943990182882538,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/max_abs":0.1357421875,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/mean":-8.137430995702744e-08,"train/train/tensor_act_model_layers_49_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/mean":5.7778379414230585e-08,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/norm":5.71875,"train/train/tensor_act_model/mean":-0.0088348388671875,"train/train/tensor_act_model_layers_34_mlp_waleed_W_g/norm":2314.1712745494747,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/max_abs":0.1455078125,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50/max_abs":23.375,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/std":6.082083303800264e-05,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/max_abs":0.0001010894775390625,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/mean":-3.743171691894531e-05,"train/train/tensor_act_model_layers_10_self_attn_q_proj/mean":-0.0595703125,"train/train/tensor_act_model_layers_43_mlp_waleed_W_g/mean":0.00475311279296875,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_6_mlp_waleed/mean":-0.0008230209350585938,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90/norm":23614.546940440196,"train/train/tensor_param_model_layers_29_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5/max_abs":26.875,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_20_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/mean":-1.1932570487260818e-07,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_mlp_waleed_W_g_weight/std":0.0240478515625,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_mlp_waleed_W_g_weight/std":5.966750542865919e-05,"train/train/tensor_act_model_layers_64_self_attn_q_proj/max_abs":5.875,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/max_abs":0.00010251998901367188,"train/train/layer__model_layers_48/param/max_abs":1,"train/train/tensor_param_model_layers_9_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_29_mlp_waleed_W_u_weight/norm":4.4375,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_57_self_attn_k_proj/mean":0.024017333984375,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/max_abs":0.0001468658447265625,"train/train/tensor_param_model_layers_57_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn_q_proj/max_abs":6.5,"train/train/tensor_act_model_layers_28_mlp_waleed/std":0.09155308148695789,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn/max_abs":1.25,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/max_abs":0.00091552734375,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/std":0.021240234375,"train/train/tensor_act_model_layers_2_self_attn_o_proj/mean":0.0013751983642578125,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_self_attn_q_proj/max_abs":7.59375,"train/train/tensor_grad_model_layers_49_mlp_waleed_W_g_weight/norm":0.014864974157341063,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/norm":0.052066243864818354,"train/train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/mean":4.00003045797348e-07,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/norm":0.00432863096695077,"train/train/tensor_grad_model_layers_50_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/mean":0.0002880096435546875,"train/train/tensor_act_model_layers_22_self_attn_o_proj/mean":-0.002796173095703125,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/norm":0.007874984016477925,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/norm":4.5625,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/mean":-4.8690708354115486e-08,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/norm":0.02035748928117101,"train/train/tensor_act_model_layers_63_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/max_abs":7.05718994140625e-05,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_down_proj/norm":446.69853263602255,"train/train/tensor_act_model_layers_42_self_attn_v_proj/mean":-0.002620697021484375,"train/train/tensor_act_model_layers_8_input_layernorm/std":1.0000000805812272,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/max_abs":7.200241088867188e-05,"train/train/layer__model_layers_31/param/std":0.048610001014254646,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/std":2.344257689747143e-05,"train/train/layer__model_layers_93/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_waleed_W_u_weight/max_abs":0.000957489013671875,"train/train/tensor_grad_model_layers_1_mlp_waleed_W_u_weight/std":0.0007611683385214832,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/max_abs":0.000957489013671875,"train/train/tensor_act_model_layers_20_self_attn_q_proj/std":1.0957083744460507,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/std":0.033447265625,"train/train/tensor_act_model_layers_41_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_84_mlp_waleed_W_u/norm":5399.690055838615,"train/train/tensor_grad_model_layers_24_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer_model_layers_36/act/std":0.8626974408809274,"train/train/tensor_act_model_layers_56_mlp_down_proj/mean":-0.0005955696105957031,"train/train/tensor_act_model_layers_88_mlp_down_proj/norm":3191.0145472387953,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/norm":4.34375,"train/train/tensor_param_model_layers_34_mlp_waleed_W_g_weight/norm":4.46875,"train/train/tensor_act_model_layers_67_mlp/std":0.12939547884792055,"train/train/tensor_act_model_layers_79_post_attention_layernorm/std":1.0000004697066092,"train/train/tensor_act_model_layers_68_mlp_down_proj/norm":661.322928585359,"train/train/tensor_param_model_layers_77_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_11_input_layernorm/max_abs":5.125,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_mlp_down_proj/std":0.09252962405123653,"train/train/tensor_act_model_layers_90_mlp_waleed_W_g/mean":-0.026763916015625,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/std":5.8695991239219123e-05,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_17_mlp_waleed_W_g_weight/mean":2.023298293352127e-07,"train/train/tensor_param_model_layers_31_mlp_waleed_W_g_weight/norm":4.46875,"train/train/tensor_act_model_layers_19_self_attn_o_proj/mean":0.00018095970153808594,"train/train/tensor_act_model_layers_35_self_attn_o_proj/norm":499.8525820415348,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/std":1.7144371341590385e-05,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/max_abs":0.0014801025390625,"train/train/tensor_act_model_layers_61_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/max_abs":0.0009002685546875,"train/train/layer__model_layers_61/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/norm":3.140625,"train/train/tensor_act_model_layers_56_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp/max_abs":0.5,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/std":0.00013211122784159693,"train/train/tensor_act_model_layers_72_mlp_down_proj/std":0.14111413048472987,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/mean":-3.759050741791725e-07,"train/train/tensor_act_model_layers_78_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp/std":0.20898439293883872,"train/train/tensor_grad_model_layers_71_mlp_waleed_W_u_weight/std":5.8056518423145925e-05,"train/train/tensor_grad_model_layers_38_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_o_proj/mean":0.00211334228515625,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/max_abs":0.000957489013671875,"train/train/tensor_act_model_layers_61_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/std":3.841445767704607e-05,"train/train/layer__model_layers_85/param/mean":0.0016389034467628706,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_0/act/mean":0.006232738494873047,"train/train/tensor_act_model_layers_71_self_attn/std":0.13647543129551903,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/std":4.774381159464339e-05,"train/train/tensor_act_model_layers_51_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_self_attn_k_proj/max_abs":4.8125,"train/train/tensor_act_model_layers_23_mlp_waleed_W_u/mean":-0.002582550048828125,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/mean":3.8673169910907745e-07,"train/train/tensor_act_model_layers_91_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_mlp_waleed/mean":-0.00011861324310302734,"train/train/tensor_grad_model_layers_31_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_q_proj/max_abs":6.6875,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/mean":0.0003795623779296875,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/std":0.026123046875,"train/train/tensor_act_model_layers_56_self_attn/std":0.12672286889455264,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/mean":2.849102020263672e-05,"train/train/tensor_act_model_layers_12_mlp/norm":409.22641123939337,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/max_abs":0.23828125,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/max_abs":0.000644683837890625,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/max_abs":0.000514984130859375,"train/train/layer__model_layers_86/param/mean":0.0017037949584389625,"train/train/layer_model_layers_59/grad/mean":-9.157981101696651e-08,"train/train/tensor_act_model_layers_44_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_input_layernorm/std":1.0000009303508013,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/max_abs":0.2392578125,"train/train/tensor_act_/max_abs":2.9462242126464844,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/std":0.04052734375,"train/train/tensor_act_model_layers_68_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/std":0.02685546875,"train/train/tensor_act_model_layers_37_mlp_waleed/mean":0.0023956298828125,"train/train/tensor_act_model_layers_45_self_attn_v_proj/mean":0.0015964508056640625,"train/train/tensor_act_model_layers_84_self_attn_o_proj/std":0.9571019647708569,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74/std":2.8242284825231185,"train/train/tensor_act_model_layers_37_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_41/grad/mean":1.3953711192433065e-07,"train/train/tensor_grad_model_layers_79_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn/mean":-0.00327301025390625,"train/train/tensor_act_model_layers_30_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_k_proj/norm":7044.48671681611,"train/train/tensor_act_model_layers_30/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_o_proj/max_abs":1.4453125,"train/train/layer__model_layers_65/param/norm":21.188634971444596,"train/train/tensor_grad_model_layers_6_mlp_waleed_W_u_weight/norm":0.04597320959267223,"train/train/tensor_act_model_layers_48_input_layernorm/std":1.0000015615939077,"train/train/tensor_act_model_layers_90_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_o_proj/mean":0.004276275634765625,"train/train/tensor_act_model_layers_59_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/layer_model_layers_92/act/norm":36694.91975091811,"train/train/tensor_act_model_layers_81_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/max_abs":0.000125885009765625,"train/train/tensor_act_model_layers_51_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/std":0,"train/train/layer__model_layers_60/param/mean":0.0014521796692180187,"train/train/tensor_act_model_layers_92_self_attn_q_proj/mean":0.031280517578125,"train/train/tensor_act_model_layers_80_self_attn_k_proj/max_abs":5.78125,"train/train/tensor_act_model_embed_tokens/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_mlp_waleed_W_g_weight/norm":0.014925498222936252,"train/train/tensor_act_model_layers_40_input_layernorm/norm":5792.610107424824,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_mlp_down_proj/mean":0.001728057861328125,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/mean":2.9453076422214508e-08,"train/train/layer__model_layers_53/param/mean":0.0015312944671107157,"train/train/tensor_act_model_layers_66_input_layernorm/norm":5792.609863289873,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/max_abs":0.1650390625,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/mean":-6.212212610989809e-08,"train/train/tensor_act_model_layers_78_self_attn_q_proj/max_abs":6.4375,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_input_layernorm/mean":0.00540924072265625,"train/train/tensor_grad_model_layers_31_mlp_waleed_W_u_weight/std":3.9639748108347864e-05,"train/train/tensor_act_model_layers_32/mean":0.0031414031982421875,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/max_abs":0.0022125244140625,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/mean":-0.00013256072998046875,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/max_abs":0.1533203125,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_waleed_W_u_weight/norm":5.03125,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_mlp_waleed_W_u_weight/mean":1.2922100722789764e-07,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp_waleed/norm":747.2725921378336,"train/train/tensor_param_model_layers_62_mlp_waleed_W_u_weight/max_abs":0.1826171875,"train/train/tensor_grad_model_embed_tokens_weight/std":0.0004083219404470758,"train/train/tensor_act_model_layers_59_mlp_waleed_W_g/norm":3038.9754154353304,"train/train/tensor_act_model_layers_92_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_q_proj/mean":-0.0064544677734375,"train/train/layer_model_layers_8/grad/max_abs":0.0030517578125,"train/train/tensor_grad_model_layers_33_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_waleed_W_u_weight/max_abs":0.1298828125,"train/train/tensor_act_model_layers_43_input_layernorm/norm":5792.606811526487,"train/train/tensor_act_model_layers_13_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/max_abs":0.1396484375,"train/train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/std":1.0908492512091487e-05,"train/train/tensor_act_model_layers_71_self_attn_k_proj/std":0.8515626181156182,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/mean":5.407491698861122e-08,"train/train/tensor_param_model_layers_67_mlp_waleed_W_g_weight/norm":5.96875,"train/train/tensor_act_model_layers_23_mlp_waleed/max_abs":2.15625,"train/train/tensor_act_model_layers_51_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/max_abs":0.000713348388671875,"train/train/tensor_act_model_layers_64_mlp_down_proj/frac_near_user_limit":0,"train/learning_rate":0.001,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn_k_proj/std":0.878906505902571,"train/train/layer__model_layers_30/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_mlp_waleed_W_g/norm":2494.643282613653,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_v_proj/mean":-0.007415771484375,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/mean":-6.258487701416016e-07,"train/train/tensor_act_model_layers_43_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_61/grad/max_abs":0.0012359619140625,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/std":0.033203125,"train/train/tensor_act_model_layers_82_mlp_waleed_W_g/max_abs":4.03125,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/norm":0.0013508230299139235,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/mean":6.437301635742188e-05,"train/train/tensor_act_model_layers_16_self_attn/max_abs":0.98828125,"train/train/tensor_act_model_layers_21_input_layernorm/mean":-0.00555419921875,"train/train/tensor_act_model_layers_57_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/std":5.116494317055934e-05,"train/train/tensor_act_model_layers_56_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_62/act/mean":0.006942287087440491,"train/train/global/param/std":0.053329693102477795,"train/train/tensor_act_model_layers_20_mlp_waleed_W_g/mean":0.00368499755859375,"train/train/tensor_act_model_layers_36_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_59/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/std":3.9458871960419336e-05,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/std":3.498842050107491e-05,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/mean":6.079673767089844e-05,"train/train/tensor_act_model_layers_48_self_attn/mean":-0.00635528564453125,"train/train/tensor_act_model_layers_41_mlp_down_proj/frac_near_user_limit":0,"train/train/layer__model_layers_37/param/mean":0.0016681355730791732,"train/train/global/grad/mean":-6.875792359580849e-08,"train/train/tensor_act_model_layers_76_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_67/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/norm":0.01876917891315576,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/std":5.080307899423908e-05,"train/train/tensor_grad_model_layers_47_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_k_proj/max_abs":6,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/layer_model_layers_59/grad/std":4.8936879527137204e-05,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/norm":0.03487559723428138,"train/train/tensor_act_model_layers_86/mean":0.0018405914306640625,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/max_abs":0.00131988525390625,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_waleed_W_g_weight/mean":-1.6676494851708412e-07,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/max_abs":0.00051116943359375,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_embed_tokens/std":0.08691406439927017,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/mean":-2.1187588572502136e-07,"train/train/tensor_act_model_layers_71/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/norm":0.0016371630077660754,"train/train/tensor_param_model_layers_60_input_layernorm_weight/mean":1,"train/train/layer_model_layers_30/act/mean":0.0009692367166280746,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_k_proj/norm":7062.749503126288,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/std":6.843493572675837e-05,"train/train/tensor_act_model_layers_5_mlp_waleed_W_u/mean":0.00020176172256469727,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_o_proj/max_abs":6.53125,"train/train/tensor_grad_model_layers_27_mlp_waleed_W_g_weight/std":4.024006597898179e-05,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/max_abs":0.000606536865234375,"train/train/tensor_param_model_layers_22_mlp_waleed_W_u_weight/mean":-0.00019931793212890625,"train/train/tensor_param_model_layers_2_mlp_waleed_W_u_weight/mean":-0.00011539459228515625,"train/train/tensor_param_model_layers_74_mlp_waleed_W_u_weight/std":0.03515625,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/max_abs":0.00043487548828125,"train/train/tensor_act_model_layers_8_self_attn/std":0.2905318737365574,"train/train/tensor_grad_model_layers_32_mlp_waleed_W_g_weight/max_abs":0.0006561279296875,"train/train/tensor_param_model_layers_59_mlp_waleed_W_g_weight/max_abs":0.15234375,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/max_abs":0.24609375,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/max_abs":0.0011749267578125,"train/train/tensor_act_model_layers_41_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17/norm":16513.248163660868,"train/train/tensor_param_model_layers_64_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/mean":5.595386028289795e-06,"train/train/tensor_act_model_layers_11_mlp_waleed_W_u/norm":1951.145057130983,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/mean":-2.930755726993084e-08,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/norm":1.688539538028167,"train/train/tensor_grad_model_layers_16_mlp_waleed_W_u_weight/std":4.396174416506025e-05,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_o_proj/mean":-0.002460479736328125,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_waleed_W_u_weight/mean":-5.359761416912079e-07,"train/train/layer_model_layers_30/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/norm":0.4511137425346442,"train/train/layer__model_layers_4/param/max_abs":1,"train/train/tensor_act_model_layers_65_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49/max_abs":23.875,"train/train/tensor_act_model_layers_75_self_attn_o_proj/max_abs":2.59375,"train/train/tensor_param_model_layers_15_mlp_waleed_W_u_weight/std":0.0230712890625,"train/train/layer_model_layers_15/act/mean":-0.0010800361633300781,"train/train/tensor_act_model_layers_32/frac_near_user_limit":0,"train/train/layer__model_layers_9/param/max_abs":1,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/max_abs":0.1328125,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/std":0.054931640625,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/std":3.731614320438381e-05,"train/train/tensor_param_model_layers_42_mlp_waleed_W_u_weight/norm":4.71875,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_post_attention_layernorm/std":1.0000000705476826,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/std":0.029296875,"train/train/tensor_act_model_layers_51_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_q_proj/std":1.156253009225177,"train/train/tensor_grad_model_layers_37_mlp_waleed_W_g_weight/max_abs":0.000804901123046875,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/std":0.0233154296875,"train/train/tensor_act_model_layers_26_post_attention_layernorm/norm":5792.192626956033,"train/train/tensor_act_model_layers_7_mlp_down_proj/mean":0.00299835205078125,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/max_abs":0.2255859375,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp/mean":0.0001709461212158203,"train/train/tensor_act_model_layers_77_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_waleed_W_u_weight/std":0.028564453125,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_72_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer_model_layers_0/grad/std":0.0032115593440624054,"train/train/tensor_act_model_layers_60_self_attn/std":0.07007228272442838,"train/train/tensor_act_model_layers_54_self_attn/mean":-0.0017547607421875,"train/train/tensor_act_model_layers_40_input_layernorm/max_abs":5.28125,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/max_abs":0.000843048095703125,"train/train/tensor_act_model_layers_25_self_attn_k_proj/std":1.0312500158042617,"train/train/tensor_act_model_layers_57_post_attention_layernorm/mean":0.0014319419860839844,"train/train/layer_model_layers_28/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_q_proj/mean":0.03338623046875,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/norm":0.004480189388715219,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/mean":-6.723403930664062e-05,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/max_abs":0.0007476806640625,"train/train/layer__model_layers_51/param/mean":0.0015399831691509849,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/norm":0.29495420973405984,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/std":0.0267333984375,"train/train/tensor_act_model_layers_17_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_mlp_waleed_W_g_weight/mean":-6.041955202817917e-08,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/norm":0.014285812579821571,"train/train/tensor_act_model_layers_50_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/max_abs":0.2275390625,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/max_abs":0.19140625,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/norm":0.018062075534043803,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/mean":-7.147900760173798e-08,"train/train/tensor_act_model_layers_20_self_attn_q_proj/norm":6365.3022829728525,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_19_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/norm":3.34375,"train/train/tensor_act_model_layers_33_mlp_waleed_W_g/max_abs":1.9765625,"train/train/tensor_act_model_layers_34_self_attn/norm":1015.0820434750545,"train/train/tensor_act_model_layers_37_input_layernorm/std":1.0000007060805054,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/mean":3.6176061257719994e-08,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/norm":0.0068980022100839505,"train/train/tensor_param_model_layers_47_mlp_waleed_W_g_weight/std":0.026611328125,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/std":0.024169921875,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90/std":4.078155432853382,"train/train/tensor_act_model_layers_30_mlp_waleed/std":0.0927734400243743,"train/train/tensor_grad_model_layers_65_mlp_waleed_W_g_weight/norm":0.01834520239336017,"train/train/layer__model_layers_31/param/mean":0.0016320119967884652,"train/train/tensor_act_model_layers_35_mlp_waleed_W_u/std":0.2812501015141423,"train/train/tensor_param_model_layers_50_mlp_waleed_W_u_weight/std":0.0272216796875,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_24/act/norm":20504.284155746554,"train/train/tensor_param_model_layers_71_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/norm":0.005914016102365688,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/max_abs":0.0007781982421875,"train/train/tensor_act_model_layers_25_self_attn_k_proj/max_abs":4.625,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/mean":7.043126970529556e-08,"train/train/tensor_act_model_layers_21_post_attention_layernorm/mean":-0.00490570068359375,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_mlp_waleed_W_u/std":0.3105469112102138,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/norm":0.02369618706452297,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/mean":-0.00028228759765625,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/std":2.738002689024887e-05,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/mean":-2.5510787963867188e-05,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_q_proj/norm":10840.272282252896,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_waleed_W_g_weight/norm":6.6875,"train/train/tensor_act_model_layers_0_mlp_waleed_W_g/std":0.9462905900753907,"train/train/tensor_act_model_layers_17_mlp_waleed_W_g/max_abs":2,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/norm":0.0014958575456924558,"train/train/tensor_act_model_layers_90_self_attn_q_proj/mean":-0.059326171875,"train/train/tensor_act_model_layers_36/mean":0.00525665283203125,"train/train/tensor_act_model_layers_85_self_attn_o_proj/max_abs":4.125,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_down_proj/norm":789.2936829990747,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/mean":-6.274785846471786e-08,"train/train/tensor_param_model_layers_68_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89/max_abs":32.5,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_5_mlp_waleed_W_g_weight/std":8.692607359247535e-05,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/std":2.987520054119791e-05,"train/train/tensor_act_model_layers_79_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/mean":-8.7738037109375e-05,"train/train/tensor_act_model_layers_67_mlp_waleed_W_g/mean":-0.001613616943359375,"train/train/layer_model_layers_17/act/max_abs":26.125,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/std":0.006595858289624231,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/norm":4.75,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/norm":0.00944769214537191,"train/train/tensor_param_model_layers_72_mlp_waleed_W_g_weight/norm":6.25,"train/train/tensor_param_model_layers_6_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_55_mlp/norm":496.48867507303925,"train/train/layer__model_layers_15/param/std":0.04735983929867052,"train/train/tensor_act_model_layers_16_mlp_waleed_W_u/norm":1997.048258581522,"train/train/tensor_param_model_layers_41_mlp_waleed_W_u_weight/norm":4.6875,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/max_abs":0.0002803802490234375,"train/train/tensor_act_model_layers_62_post_attention_layernorm/norm":5792.606811524923,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/max_abs":0.00019073486328125,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/mean":-2.5987625122070312e-05,"train/train/tensor_act_model_layers_55_self_attn/std":0.06543248470708976,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/max_abs":0.000576019287109375,"train/train/tensor_act_model_layers_28_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_k_proj/std":1.1328125867350314,"train/train/tensor_act_model_layers_7/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp_waleed_W_u/std":0.25585941276477214,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_mlp_waleed_W_g_weight/max_abs":0.0008087158203125,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/norm":5.0625,"train/train/layer_model_layers_22/act/std":0.87733719030871,"train/train/tensor_act_model_layers_8_input_layernorm/max_abs":4.96875,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/norm":5.90625,"train/train/tensor_param_model_layers_27_input_layernorm_weight/std":0.00103759765625,"train/train/layer_model_layers_75/act/mean":0.008945822715759277,"train/train/tensor_act_model_layers_80_mlp_waleed_W_g/mean":0.012542724609375,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/mean":-5.8178557083010674e-08,"train/train/tensor_act_model_layers_11_self_attn_q_proj/max_abs":8.4375,"train/train/layer__model_layers_65/param/mean":0.0016023989958621782,"train/train/tensor_act_model_layers_31_input_layernorm/norm":5792.610961921413,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp_waleed/norm":6422.5897013451995,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/std":4.2861859440335e-05,"train/train/tensor_act_model_layers_39_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_v_proj/std":0.43212974973908774,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/max_abs":0.1708984375,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/mean":6.952905096113682e-08,"train/train/tensor_param_model_layers_55_mlp_waleed_W_u_weight/max_abs":0.1455078125,"train/train/tensor_param_model_layers_13_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_41/act/std":0.8180421246790314,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/mean":0.000255584716796875,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/max_abs":0.2373046875,"train/train/tensor_act_model_layers_44_self_attn/norm":328.80636347516275,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_post_attention_layernorm/mean":-0.0113525390625,"train/train/tensor_act_model_layers_93_self_attn_q_proj/norm":7151.30399476778,"train/train/tensor_act_model_layers_12_self_attn_q_proj/max_abs":5.1875,"train/train/layer_model_layers_15/grad/max_abs":0.0020904541015625,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/max_abs":0.000133514404296875,"train/train/tensor_act_model_layers_50_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/norm":4.71875,"train/train/tensor_act_model_layers_82_self_attn_q_proj/std":1.1777397142988293,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/std":5.882344442946641e-05,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/std":1.2110286068932973e-05,"train/train/tensor_param_model_layers_46_mlp_waleed_W_g_weight/max_abs":0.1826171875,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/max_abs":0.123046875,"train/train/tensor_act_model_layers_92_mlp_down_proj/mean":-0.0302734375,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/norm":4.53125,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/max_abs":0.2060546875,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/norm":6.3125,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_v_proj/std":0.3457031368493695,"train/train/tensor_act_model_layers_35_post_attention_layernorm/max_abs":5.46875,"train/train/tensor_param_model_layers_34_mlp_waleed_W_g_weight/max_abs":0.1494140625,"train/train/tensor_act_model_layers_92_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_46/act/mean":-0.0014601945877075195,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/std":0.031005859375,"train/train/tensor_act_model_layers_1/max_abs":27.375,"train/train/tensor_act_model_layers_58/std":2.593775187990443,"train/train/tensor_act_model_layers_87_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/norm":0.0006681113108401467,"train/train/layer_model_layers_74/grad/std":9.326378616422112e-05,"train/train/tensor_act_model_layers_91_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_33/grad/max_abs":0.000598907470703125,"train/train/layer__model_layers_33/param/mean":0.0015942935080684477,"train/train/tensor_act_model_layers_31_self_attn_o_proj/max_abs":1.90625,"train/train/tensor_param_model_layers_26_input_layernorm_weight/mean":1,"train/train/layer__model_layers_45/param/std":0.04927804524422577,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/norm":0.005458339932913346,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/std":0.040771484375,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/mean":-0.0004863739013671875,"train/train/tensor_act_model_layers_59_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/max_abs":0.000885009765625,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/std":0.00018665604394656318,"train/train/tensor_act_model_layers_36_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_q_proj/max_abs":6.71875,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/norm":0.013614465889087305,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/std":5.241304198716471e-05,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_83_mlp_waleed_W_g_weight/norm":0.03021542312470992,"train/train/tensor_act_model_layers_82_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_waleed_W_u/mean":-0.01202392578125,"train/train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/std":6.109914287528745e-05,"train/train/layer_model_layers_72/act/mean":0.006569564342498779,"train/train/tensor_act_model_layers_76/max_abs":26,"train/train/tensor_act_model_layers_40_self_attn_k_proj/max_abs":4.71875,"train/train/tensor_param_model_layers_92_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_mlp_waleed_W_g_weight/std":0.034423828125,"train/train/tensor_param_model_layers_8_mlp_waleed_W_u_weight/std":0.0228271484375,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/std":0.052001953125,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/max_abs":0.1474609375,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_74/grad/mean":1.8569079261162725e-07,"train/train/tensor_grad_model_layers_5_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/std":6.106268066699162e-05,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_64_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_70/grad/max_abs":0.00194549560546875,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/mean":4.522502422332764e-06,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/mean":-8.569622877985239e-08,"train/train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_mlp_waleed_W_u_weight/mean":-5.273614078760147e-08,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/max_abs":0.19140625,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_mlp_waleed_W_g_weight/std":0.0299072265625,"train/train/tensor_param_model_layers_63_mlp_waleed_W_u_weight/mean":-0.00010395050048828125,"train/train/tensor_act_model_layers_70_post_attention_layernorm/mean":0.00402069091796875,"train/train/tensor_act_model_layers_40_mlp_waleed/std":0.10986329731634804,"train/train/tensor_act_model_layers_49_self_attn_q_proj/max_abs":4.4375,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/norm":6.90625,"train/train/tensor_param_model_layers_5_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_24_mlp_waleed_W_u/max_abs":2.65625,"train/train/tensor_act_model_embed_tokens/mean":-0.00044345855712890625,"train/train/tensor_param_model_layers_84_mlp_waleed_W_g_weight/std":0.041748046875,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/std":8.393224362072808e-05,"train/train/tensor_act_model_layers_57_self_attn_k_proj/norm":5615.549373896205,"train/train/tensor_grad_model_layers_79_mlp_waleed_W_u_weight/max_abs":0.00102996826171875,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/std":0.0008145383709485915,"train/train/tensor_param_model_layers_69_mlp_waleed_W_g_weight/norm":6,"train/train/tensor_act_model_layers_80_self_attn_q_proj/mean":-0.0345458984375,"train/train/tensor_act_model_layers_0_mlp_waleed_W_u/norm":9196.944352681627,"train/train/tensor_act_model_layers_58_self_attn_k_proj/std":1.0390625430229008,"train/train/tensor_grad_model_layers_91_mlp_waleed_W_g_weight/max_abs":0.00174713134765625,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_embed_tokens_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/max_abs":5.030632019042969e-05,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/mean":1.941225491464138e-08,"train/train/tensor_act_model_layers_89_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/norm":0.01776217428629557,"train/train/tensor_act_model_layers_17_self_attn_v_proj/std":0.4335938369905539,"train/train/tensor_act_model_layers_82_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/max_abs":0.1259765625,"train/train/tensor_param_model_layers_67_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_9_mlp_down_proj/mean":-0.002750396728515625,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/mean":-4.4284388422966003e-07,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/std":0.03369140625,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/norm":0.0017752119067227872,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/mean":6.737536750733852e-08,"train/train/tensor_act_model_layers_47_mlp_waleed/mean":1.5527009963989258e-05,"train/train/tensor_act_model_layers_21_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/norm":4.9375,"train/train/tensor_act_model_layers_38_mlp_down_proj/norm":311.7933821182262,"train/train/layer_model_layers_24/grad/mean":-1.631536217356993e-07,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_mlp_waleed_W_u_weight/norm":4.28125,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/max_abs":8.0108642578125e-05,"train/train/tensor_act_model_layers_27_self_attn_v_proj/mean":0.00750732421875,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/norm":0.0028022008979637024,"train/train/tensor_grad_model_layers_62_mlp_waleed_W_g_weight/std":4.7094494895500394e-05,"train/train/tensor_act_model_layers_81_self_attn/norm":2964.565964577804,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/std":0.024169921875,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_13_input_layernorm/mean":-0.00601959228515625,"train/train/tensor_act_model_layers_50_mlp_waleed/mean":-0.0015544891357421875,"train/train/tensor_param_model_layers_63_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_mlp_waleed_W_g_weight/max_abs":0.00106048583984375,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/mean":-3.0442606657743454e-08,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_40/grad/norm":0.030018516387880778,"train/train/tensor_act_model_layers_88_mlp/norm":3191.0145472387953,"train/train/tensor_act_model_layers_0_input_layernorm/std":1.000000026862835,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/norm":0.000561917337874452,"train/train/tensor_act_model_layers_90_self_attn/max_abs":10.8125,"train/train/tensor_act_model_layers_85_self_attn/norm":2511.450679147774,"train/train/tensor_act_model_layers_32_self_attn_k_proj/std":0.809572090427669,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/max_abs":0.25390625,"train/train/tensor_param_model_layers_14_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_37_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_mlp_down_proj/mean":-0.00017070770263671875,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_input_layernorm/norm":5792.301025392811,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/max_abs":0.00074005126953125,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_mlp_waleed_W_g_weight/norm":0.014375172282306075,"train/train/tensor_param_model_layers_14_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_41_mlp_down_proj/std":0.06542971819193676,"train/train/tensor_grad_model_layers_13_mlp_waleed_W_g_weight/std":6.891369816568254e-05,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_51/act/max_abs":23.375,"train/train/tensor_act_model_layers_9_self_attn_k_proj/max_abs":5.1875,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_o_proj/mean":-0.00347137451171875,"train/train/tensor_param_model_layers_59_mlp_waleed_W_u_weight/norm":5.34375,"train/train/tensor_act_model_layers_38_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/std":0.033935546875,"train/train/tensor_act_model_layers_56_mlp_waleed/mean":0.0026397705078125,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/std":3.276008098092373e-05,"train/train/layer__model_layers_35/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/norm":0.01834195304660322,"train/train/layer__model_layers_11/param/mean":0.0016732416733192764,"train/train/tensor_param_model_layers_28_mlp_waleed_W_g_weight/std":0.0244140625,"train/train/tensor_act_model_layers_87_mlp_waleed/std":0.5004930626245383,"train/train/tensor_param_model_layers_54_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/std":4.470500590616677e-05,"train/train/tensor_act_model_layers_13/frac_near_user_limit":0,"train/train/layer__model_layers_9/param/mean":0.0015793932767442533,"train/train/tensor_act_model_layers_2_mlp_down_proj/max_abs":3.890625,"train/train/tensor_act_model_layers_76_input_layernorm/std":1.0000007169726062,"train/train/tensor_grad_model_layers_13_mlp_waleed_W_g_weight/mean":-6.170012056827545e-08,"train/train/tensor_act_model_layers_77_post_attention_layernorm/mean":0.00739288330078125,"train/train/tensor_act_model_layers_57_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp_waleed/max_abs":4.09375,"train/train/tensor_act_model_layers_14_self_attn/mean":-0.0001175999641418457,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/std":0.0400390625,"train/train/layer_model_layers_52/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/mean":1.0970979928970337e-06,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/mean":-1.0993098840117455e-06,"train/train/tensor_act_model_layers_24_self_attn_q_proj/mean":0.0124664306640625,"train/train/tensor_act_model_layers_82_input_layernorm/norm":5792.613281253838,"train/train/tensor_grad_model_layers_74_mlp_waleed_W_g_weight/max_abs":0.001068115234375,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/norm":4.75,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/max_abs":0.0001888275146484375,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/mean":-7.613562047481537e-08,"train/train/tensor_param_model_layers_32_mlp_waleed_W_g_weight/std":0.0247802734375,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_4_mlp_waleed_W_g_weight/mean":-1.2993812561035156e-05,"train/train/layer_model_layers_81/act/std":1.0577228737684774,"train/train/tensor_act_model_layers_55_mlp_waleed_W_g/std":0.36767683877577556,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/norm":0.011830827094457947,"train/train/tensor_act_model_layers_68_self_attn/norm":813.8348934941843,"train/train/tensor_act_model_layers_6_self_attn_k_proj/norm":10052.122653825027,"train/train/tensor_act_model_layers_91_self_attn_v_proj/mean":0.027191162109375,"train/train/tensor_param_model_layers_33_mlp_waleed_W_u_weight/max_abs":0.10595703125,"train/train/tensor_act_model_layers_14_self_attn/max_abs":0.88671875,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/mean":0.000194549560546875,"train/train/tensor_act_model_layers_11_mlp/norm":352.1878185351718,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/norm":0.0005721759788989663,"train/train/tensor_act_model_layers_93_self_attn_v_proj/std":0.7607494838692791,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/norm":4.625,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_k_proj/mean":0.03729248046875,"train/train/tensor_act_model_layers_28_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_waleed_W_g_weight/mean":-1.1641532182693481e-10,"train/train/layer__model_layers_66/param/std":0.053317187406698816,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/mean":0.0004138946533203125,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_38_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_mlp_waleed_W_u_weight/mean":5.384208634495735e-08,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_31/grad/max_abs":0.00164031982421875,"train/train/tensor_act_model_layers_62_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/norm":0.05207854884742283,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/std":0.0255126953125,"train/train/tensor_act_model_layers_41/max_abs":25,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/mean":8.64383764564991e-08,"train/train/tensor_act_model_layers_91_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71/max_abs":25.25,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/norm":0.01462362094637324,"train/train/layer_model_layers_44/act/norm":19200.495364219514,"train/train/tensor_act_model_layers_22_mlp_down_proj/max_abs":0.6640625,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/std":2.69083813359701e-05,"train/train/tensor_param_model_layers_41_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_62_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_mlp_down_proj/norm":647.1625773204828,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn/max_abs":1.109375,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/max_abs":0.00014400482177734375,"train/train/tensor_act_model_layers_0_mlp_waleed/frac_near_user_limit":0,"train/train/layer_model_layers_12/grad/std":5.822436566663299e-05,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/mean":-1.1723022907972336e-07,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/std":0.031494140625,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/max_abs":0.2001953125,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/norm":4.46875,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_mlp_waleed_W_u_weight/norm":4.15625,"train/train/tensor_grad_model_layers_15_mlp_waleed_W_u_weight/std":5.589416129148518e-05,"train/train/tensor_param_model_layers_69_input_layernorm_weight/std":0,"train/train/layer_model_layers_68/act/max_abs":25,"train/train/tensor_act_model_layers_45_self_attn_o_proj/mean":-0.00096893310546875,"train/train/tensor_param_model_layers_90_mlp_waleed_W_u_weight/mean":0.00022029876708984375,"train/train/tensor_act_model_layers_41_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/std":0.057861328125,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/mean":-0.00021457672119140625,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/max_abs":0.208984375,"train/train/tensor_act_model_layers_19/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_mlp_waleed_W_u_weight/mean":1.4066696166992188e-05,"train/train/tensor_act_model_layers_82_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_waleed_W_g_weight/std":0.0291748046875,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/std":0.0001268810556830821,"train/train/tensor_act_model_layers_10/mean":-0.00667572021484375,"train/train/tensor_act_model_layers_30_self_attn_v_proj/mean":0.0139007568359375,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/max_abs":0.00022029876708984375,"train/train/tensor_act_model_layers_19/max_abs":26.125,"train/train/tensor_act_model_layers_45/std":2.625025550533091,"train/train/tensor_act_model_layers_80_self_attn/norm":3130.742206041148,"train/train/tensor_act_model_layers_46_mlp_waleed_W_g/norm":2643.212513586525,"train/train/tensor_grad_model_layers_14_mlp_waleed_W_u_weight/norm":0.014024910756454491,"train/train/tensor_act_model_layers_46_self_attn_k_proj/std":0.866212625512647,"train/train/tensor_act_model_layers_23/norm":16253.776817670518,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/norm":0.004082724190719365,"train/train/tensor_grad_model_layers_28_mlp_waleed_W_g_weight/std":3.840133029247284e-05,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/max_abs":0.1298828125,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/std":3.846192467124756e-05,"train/train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/std":4.804271589301922e-05,"train/train/tensor_grad_model_layers_74_mlp_waleed_W_u_weight/std":6.086611411939855e-05,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/mean":-3.5064294934272766e-07,"train/train/tensor_act_model_layers_36_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/norm":4.8125,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/mean":-4.673004150390625e-05,"train/train/tensor_grad_model_layers_0_mlp_waleed_W_u_weight/norm":0.539211086858761,"train/train/layer__model_layers_88/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/max_abs":0.1953125,"train/train/tensor_param_model_layers_24_mlp_waleed_W_u_weight/max_abs":0.1181640625,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/max_abs":0.00150299072265625,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/norm":0.010651115690811955,"train/train/tensor_act_model_layers_0_mlp_down_proj/max_abs":10.9375,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/norm":0.002615680919269373,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/std":3.259303979335995e-05,"train/train/tensor_act_model_layers_31_self_attn_v_proj/std":0.45263751302781113,"train/train/tensor_act_model_layers_91_self_attn_o_proj/std":0.6250062124844148,"train/train/tensor_act_model_layers_14_post_attention_layernorm/mean":-0.00562286376953125,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/std":0.042236328125,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_waleed_W_g_weight/std":0.028076171875,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/norm":0.0016369674491961085,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/mean":-4.655122756958008e-05,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/mean":2.474989742040634e-07,"train/train/tensor_act_model_layers_64_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/norm":0.012105712028981138,"train/train/tensor_param_model_layers_46_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_46_self_attn_o_proj/norm":512.7288572803968,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_57_post_attention_layernorm/std":1.0000009575531938,"train/train/tensor_grad_model_layers_34_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp_waleed_W_u/max_abs":2.75,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/std":0.0283203125,"train/train/layer__model_layers_72/param/std":0.05477276802202651,"train/train/tensor_act_model_layers_13_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn_v_proj/mean":0.001956939697265625,"train/train/tensor_act_model_layers_72_mlp_waleed_W_u/max_abs":3.40625,"train/train/tensor_act_model_layers_26_input_layernorm/norm":5792.607299806606,"train/train/tensor_act_model_layers_51_mlp_waleed_W_u/mean":0.00015306472778320312,"train/train/tensor_grad_model_layers_50_mlp_waleed_W_g_weight/norm":0.015921316204429065,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/norm":4.3125,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_o_proj/norm":629.7152131728639,"train/train/tensor_act_model_layers_5_mlp_down_proj/norm":913.1975908258453,"train/train/tensor_param_model_layers_82_input_layernorm_weight/std":0,"train/train/layer__model_layers_78/param/norm":22.971981065256433,"train/train/layer_model_layers_22/act/mean":-7.909536361694336e-05,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/mean":-1.1420343071222305e-07,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/mean":-0.00010585784912109375,"train/train/tensor_act_model_layers_21_self_attn_k_proj/mean":-0.033935546875,"train/train/tensor_param_model_layers_81_mlp_waleed_W_g_weight/norm":7.28125,"train/train/tensor_param_model_layers_83_mlp_waleed_W_u_weight/norm":7.4375,"train/train/tensor_act_model_layers_22_self_attn_v_proj/mean":0.003879547119140625,"train/train/tensor_act_model_layers_24_input_layernorm/max_abs":5.25,"train/train/tensor_act_model_layers_12_self_attn_q_proj/mean":0.00348663330078125,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/norm":0.041998929214451676,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/std":1.4911075001101487e-05,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/norm":4.25,"train/train/tensor_act_model_layers_79_self_attn/std":0.30127392107565254,"train/train/tensor_param_model_layers_59_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_60_mlp_waleed_W_u_weight/std":0.029052734375,"train/train/tensor_param_model_layers_81_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_1_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp_down_proj/std":0.07409698298810914,"train/train/tensor_act_model_layers_72_mlp_waleed_W_g/norm":3696.2974532397093,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/norm":0.00336013987616368,"train/train/tensor_act_model_layers_19_self_attn_v_proj/max_abs":1.421875,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/max_abs":0.00012874603271484375,"train/train/tensor_act_model_layers_55_self_attn/max_abs":0.7734375,"train/train/tensor_act_model_layers_18_mlp/max_abs":0.50390625,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/max_abs":0.000385284423828125,"train/train/tensor_param_model_layers_93_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_39/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/norm":0.0243628919312535,"train/train/tensor_act_model_layers_27_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_42_self_attn_o_proj/mean":0.0008697509765625,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/std":0.0245361328125,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/std":3.250273911580048e-05,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/norm":5.03125,"train/train/tensor_param_model_layers_21_input_layernorm_weight/std":0,"train/train/layer_model_layers_73/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_v_proj/norm":2233.4626429479235,"train/train/tensor_act_model_layers_54_mlp_waleed/norm":1003.7434693987797,"train/train/tensor_param_model_layers_16_mlp_waleed_W_g_weight/max_abs":0.11083984375,"train/train/layer__model_layers_31/param/max_abs":1,"train/train/tensor_act_model_layers_23_self_attn_q_proj/mean":0.022552490234375,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_mlp_waleed_W_u_weight/mean":8.320785127580166e-08,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/norm":0.002011908370636633,"train/train/tensor_act_model_layers_10_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_k_proj/norm":6237.188737920434,"train/train/tensor_grad_model_layers_2_mlp_waleed_W_g_weight/max_abs":0.004058837890625,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/norm":4.28125,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/mean":-2.986053004860878e-08,"train/train/tensor_act_model_layers_4_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/max_abs":0.00494384765625,"train/train/tensor_act_model_layers_76_mlp_waleed/mean":0.0011997222900390625,"train/train/tensor_act_model_layers_62_self_attn_o_proj/norm":507.8259041262101,"train/train/tensor_act_model_layers_49_mlp_waleed/norm":1057.3288459442,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/std":0.03271484375,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_waleed_W_g_weight/std":0.0252685546875,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_mlp_waleed_W_u_weight/max_abs":0.000568389892578125,"train/train/layer_model_layers_34/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/max_abs":0.240234375,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/std":0.0390625,"train/train/tensor_act_model_layers_53_mlp_waleed/norm":1034.1938061520666,"train/train/tensor_grad_model_layers_8_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67/norm":15518.701455538636,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/mean":-6.609479896724224e-08,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/max_abs":0.000438690185546875,"train/train/layer_model_layers_38/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/norm":6.09375,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/max_abs":0.00016880035400390625,"train/train/tensor_act_model_layers_34_self_attn_v_proj/mean":-0.001476287841796875,"train/train/tensor_param_model_layers_66_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/mean":1.1210795491933823e-07,"train/train/tensor_act_model_layers_87_self_attn_v_proj/std":0.6669954199314683,"train/train/tensor_act_model_layers_15_self_attn_v_proj/norm":2113.0235108347,"train/train/tensor_act_model_layers_52_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_input_layernorm/std":1.0000000711297592,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/mean":-0.0003414154052734375,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/max_abs":0.0014190673828125,"train/train/tensor_act_model_layers_21/std":2.773492754558219,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/mean":-7.724761962890625e-05,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/std":0.00011290358077379729,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/norm":0.018192572420538837,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/mean":-2.3283064365386963e-07,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/std":8.056628303641035e-05,"train/train/tensor_act_model_layers_89_mlp_waleed_W_g/mean":-0.0382080078125,"train/train/tensor_act_model_layers_14_mlp/std":0.06921419466199902,"train/train/tensor_act_model_layers_40_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/max_abs":0.279296875,"train/train/layer_model_layers_64/act/max_abs":24.5,"train/train/tensor_act_model_layers_38_mlp/mean":-5.8025121688842773e-05,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/max_abs":0.00015544891357421875,"train/train/tensor_act_model_layers_64_self_attn_k_proj/std":0.9443374626873209,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/max_abs":0.00555419921875,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn/max_abs":2.6875,"train/train/layer_model_layers_58/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/norm":0.005945016908753927,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/norm":5.71875,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/std":0.00016380246124140044,"train/train/tensor_act_model_layers_58_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_mlp_waleed_W_u_weight/mean":-1.494772732257843e-07,"train/train/tensor_param_model_layers_18_mlp_waleed_W_u_weight/max_abs":0.11376953125,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/mean":4.4761691242456436e-08,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/mean":3.189779818058014e-07,"train/train/tensor_act_model_layers_86_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn/std":0.08496171841038999,"train/train/tensor_act_model_layers_14_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/max_abs":0.000244140625,"train/train/tensor_act_model_layers_52_mlp_waleed_W_g/mean":-0.002582550048828125,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/std":6.094956835424663e-05,"train/train/layer_model_layers_27/grad/norm":0.043991873458272414,"train/train/tensor_param_model_layers_24_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_self_attn_q_proj/max_abs":5.40625,"train/train/tensor_act_model_layers_40_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn_k_proj/std":0.9960937531263221,"train/train/tensor_grad_model_layers_49_mlp_waleed_W_g_weight/max_abs":0.0004863739013671875,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/norm":0.0007747686771999321,"train/train/tensor_act_model_layers_56_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_36/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_mlp_waleed_W_u_weight/norm":5.09375,"train/train/tensor_param_model_layers_72_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_waleed_W_g_weight/max_abs":0.169921875,"train/train/tensor_act_model_layers_60_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/std":3.6062575650209305e-05,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/max_abs":0.001739501953125,"train/train/tensor_act_model_layers_77_self_attn/std":0.3515652745852656,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/max_abs":0.0005645751953125,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/max_abs":0.00020503997802734375,"train/train/layer_model_layers_66/grad/norm":0.04416927300350204,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_waleed_W_u/mean":0.0003437995910644531,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/mean":-1.055002212524414e-05,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/mean":-0.00017452239990234375,"train/train/layer_model_layers_16/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_post_attention_layernorm/norm":5792.605834965366,"train/train/tensor_param_model_layers_61_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/norm":7.65625,"train/train/tensor_param_model_layers_1_mlp_waleed_W_u_weight/max_abs":0.15625,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/std":5.88488418091057e-05,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/max_abs":0.0001239776611328125,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/std":0.03759765625,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp_waleed_W_u/norm":2812.8373712596594,"train/train/layer__model_layers_51/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_87/param/mean":0.0014602732546801873,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/norm":0.005829676226081263,"train/train/tensor_act_model_layers_30_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/std":0.04541015625,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/norm":4.46875,"train/train/tensor_act_model_layers_79_self_attn_k_proj/max_abs":6.03125,"train/train/tensor_act_model_layers_83_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/norm":0.026801175809445927,"train/train/tensor_act_model_layers_56_input_layernorm/mean":0.0008387565612792969,"train/train/tensor_act_model_layers_19_mlp/max_abs":0.462890625,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/norm":4.78125,"train/train/tensor_act_model_layers_86_self_attn/std":0.3530609017134962,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/norm":3.078125,"train/train/tensor_act_model_layers_0_self_attn_q_proj/mean":-0.0110626220703125,"train/train/tensor_grad_model_layers_9_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_o_proj/mean":-0.0018863677978515625,"train/train/tensor_param_model_layers_91_mlp_waleed_W_u_weight/mean":-5.1975250244140625e-05,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/max_abs":0.11865234375,"train/train/tensor_act_model_layers_52_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/std":0.00011075108657649735,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/std":4.101874189307112e-05,"train/train/tensor_act_model_layers_73_post_attention_layernorm/max_abs":5.15625,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21/norm":16057.124456336127,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/mean":0.00014400482177734375,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/std":0.025146484375,"train/train/tensor_act_model_layers_20_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_v_proj/mean":0.013336181640625,"train/train/tensor_act_model_layers_38_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn_k_proj/std":1.2578141941035454,"train/train/tensor_grad_model_layers_25_mlp_waleed_W_u_weight/std":3.981911733577975e-05,"train/train/tensor_act_model_layers_76_mlp_waleed_W_u/mean":-0.00273895263671875,"train/train/tensor_act_model_layers_34_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/mean":0.00013828277587890625,"train/train/tensor_param_model_layers_54_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/max_abs":0.001007080078125,"train/train/tensor_act_model_layers_10_mlp_waleed/norm":1301.0976324028434,"train/train/layer_model_layers_29/grad/norm":0.03424864708519923,"train/train/tensor_act_model_layers_1_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp_waleed_W_u/norm":3480.3365607826872,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/norm":0.000705474089321629,"train/train/tensor_act_model_layers_78_input_layernorm/max_abs":5.46875,"train/train/tensor_act_model_layers_67_input_layernorm/norm":5792.615600588462,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_mlp_waleed/norm":965.7208416843118,"train/train/tensor_act_model_layers_25_mlp/norm":405.5003717256925,"train/train/tensor_act_model_layers_74_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/max_abs":0.255859375,"train/train/tensor_act_model_layers_76_mlp_waleed/std":0.23535164216091414,"train/train/tensor_act_model_layers_30_self_attn_o_proj/std":0.14649581924079685,"train/train/tensor_act_model_layers_71_self_attn_q_proj/max_abs":5.15625,"train/train/layer_model_layers_76/grad/max_abs":0.00124359130859375,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/norm":0.022066129439946295,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp_waleed_W_g/norm":2315.2600176582537,"train/train/layer__model_layers_91/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/mean":-4.982948303222656e-05,"train/train/tensor_act_model_layers_10_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/max_abs":0.15234375,"train/train/tensor_act_model_layers_15_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/max_abs":0.000804901123046875,"train/train/tensor_param_model_layers_8_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp_down_proj/max_abs":1.078125,"train/train/tensor_param_model_layers_72_mlp_waleed_W_g_weight/std":0.034423828125,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/mean":0.000164031982421875,"train/train/tensor_act_model_layers_40_self_attn_v_proj/norm":1877.7793466961018,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/max_abs":0.00125885009765625,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/norm":3.09375,"train/train/tensor_param_model_layers_3_mlp_waleed_W_g_weight/std":0.024169921875,"train/train/layer_model_layers_2/act/std":1.041285039605954,"train/train/layer_model_layers_71/grad/std":5.520336263315647e-05,"train/train/tensor_act_model_layers_27_mlp_waleed/norm":786.3902812473624,"train/train/tensor_act_model_layers_75_self_attn/std":0.2607441840299499,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_waleed_W_u_weight/max_abs":0.002960205078125,"train/train/tensor_act_model_layers_7_self_attn_q_proj/mean":0.00460052490234375,"train/train/tensor_param_model_embed_tokens_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/norm":0.0033196371457524503,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/std":0.048095703125,"train/train/tensor_act_model_layers_20_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/mean":-1.8079299479722977e-07,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/mean":1.329183578491211e-05,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/norm":3.375,"train/train/layer__model_layers_29/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp_waleed_W_g/max_abs":2.734375,"train/train/tensor_act_model_layers_35_mlp/max_abs":0.5625,"train/train/tensor_param_model_norm_weight/max_abs":1,"train/train/tensor_act_model_layers_78_mlp/std":0.20166084274857496,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/mean":-6.723403930664062e-05,"train/train/layer_model_layers_1/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/std":9.51675772148503e-05,"train/train/tensor_act_model_layers_21_mlp_waleed_W_g/max_abs":2.171875,"train/train/tensor_act_model_layers_17_self_attn_q_proj/max_abs":6.1875,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/max_abs":0.000965118408203125,"train/train/tensor_act_model_layers_63_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_mlp_waleed_W_u_weight/std":9.586539307808923e-05,"train/train/tensor_act_model_layers_85/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_input_layernorm/norm":5792.613037111768,"train/train/tensor_act_model_layers_48_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_24/grad/max_abs":0.002410888671875,"train/train/tensor_grad_model_layers_19_mlp_waleed_W_u_weight/std":4.906690533619632e-05,"train/train/tensor_act_model_layers_17_mlp_waleed_W_u/std":0.2656250101623726,"train/train/tensor_act_model_layers_25_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/std":0.037353515625,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/std":1.094613072012211e-05,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/max_abs":0.298828125,"train/train/layer__model_layers_11/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp_waleed_W_u/norm":2184.8538674657457,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/norm":0.01953372939584912,"train/train/tensor_param_model_layers_30_mlp_waleed_W_g_weight/std":0.0247802734375,"train/train/tensor_act_model_layers_36_self_attn_v_proj/mean":-0.00830078125,"train/train/tensor_act_model_layers_78_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_mlp_waleed_W_u_weight/max_abs":0.00121307373046875,"train/train/tensor_param_model_layers_69_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_83_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/std":0.04052734375,"train/train/tensor_act_model_layers_9_self_attn_k_proj/mean":-0.0169677734375,"train/train/tensor_grad_model_layers_34_mlp_waleed_W_g_weight/norm":0.013384767626700036,"train/train/tensor_act_model_layers_18_self_attn/std":0.10351739021326295,"train/train/layer_model_layers_13/grad/mean":2.3046238188066646e-08,"train/train/tensor_act_model_layers_34_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn/max_abs":2.578125,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/max_abs":0.2119140625,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/norm":0.021211890239639854,"train/train/tensor_param_model_layers_67_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/max_abs":0.0009918212890625,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_64/act/std":0.8641197833619536,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/max_abs":0.00074005126953125,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_57/grad/std":4.812236760557168e-05,"train/train/tensor_act_model_layers_37_mlp_down_proj/max_abs":0.69140625,"train/train/layer_model_layers_91/act/max_abs":36,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/max_abs":0.000766754150390625,"train/train/tensor_act_model_layers_45_self_attn_v_proj/std":0.4082032065063135,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/mean":1.3057142496109009e-06,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/mean":0.0002193450927734375,"train/train/tensor_act_model_layers_29_mlp/max_abs":0.404296875,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_mlp_waleed_W_g_weight/mean":-5.751848220825195e-06,"train/train/tensor_act_model_layers_58_self_attn_o_proj/norm":1144.703296227091,"train/train/tensor_act_model_layers_47_self_attn_o_proj/std":0.027649560878366185,"train/train/tensor_grad_model_layers_57_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_31/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp_down_proj/mean":-7.243454456329346e-05,"train/train/tensor_act_model_layers_20_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/mean":-4.159119271207601e-08,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/norm":0.026624903584695242,"train/train/tensor_act_model_layers_66_mlp_waleed_W_g/mean":0.00466156005859375,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/norm":4.34375,"train/train/tensor_act_model_layers_86_mlp_waleed_W_g/mean":-5.340576171875e-05,"train/train/layer_model_layers_59/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/std":1.6908724283392083e-05,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_post_attention_layernorm/mean":0.0039196014404296875,"train/train/tensor_act_model_layers_39_self_attn_v_proj/max_abs":2.890625,"train/train/tensor_act_model_layers_85_self_attn_v_proj/std":0.5937500862698744,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_self_attn_o_proj/norm":902.5659256278218,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/std":9.657229890587028e-05,"train/train/tensor_param_model_layers_37_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_47_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/std":3.3125285313934105e-05,"train/train/tensor_grad_model_layers_11_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/max_abs":0.0013275146484375,"train/train/tensor_param_model_layers_78_mlp_waleed_W_g_weight/norm":6.84375,"train/train/tensor_act_model_layers_2_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp_waleed_W_u/max_abs":2.125,"train/train/tensor_act_model_layers_35_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_38_mlp_waleed_W_u_weight/max_abs":0.00069427490234375,"train/train/tensor_act_model_layers_30_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_mlp_waleed_W_u_weight/max_abs":0.00089263916015625,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_36/grad/mean":2.6787147390861035e-08,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/mean":0.00016880035400390625,"train/train/tensor_act_model_layers_89_mlp_waleed/mean":-0.001392364501953125,"train/train/layer__model_layers_89/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/norm":0.008908885515669493,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/std":0.0220947265625,"train/train/tensor_act_model_layers_31_self_attn_o_proj/norm":872.910595406355,"train/train/tensor_act_model_layers_50_input_layernorm/norm":5792.609863284121,"train/train/tensor_act_model_layers_32_self_attn_o_proj/mean":-0.0002598762512207031,"train/train/layer__model_layers_90/param/max_abs":1,"train/train/tensor_act_model_layers_73_post_attention_layernorm/mean":-0.0005002021789550781,"train/train/tensor_param_model_layers_51_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/std":0.055419921875,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/max_abs":0.166015625,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_61_self_attn_q_proj/mean":0.0124969482421875,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/std":0.00015959826927573604,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_83/act/mean":0.02329949289560318,"train/train/tensor_grad_model_layers_6_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_q_proj/std":1.0234375636996185,"train/train/tensor_act_model_layers_30_mlp_waleed_W_g/norm":2279.403857707389,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_81_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_waleed/std":0.09558129209714883,"train/train/layer_model_layers_35/grad/std":3.7820025344231515e-05,"train/train/tensor_grad_model_layers_36_mlp_waleed_W_g_weight/mean":7.916241884231567e-08,"train/train/tensor_grad_model_layers_20_mlp_waleed_W_u_weight/mean":2.3562461137771606e-07,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/max_abs":0.1201171875,"train/train/tensor_param_model_layers_51_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_67_post_attention_layernorm/max_abs":5.125,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_17/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp_waleed/std":0.34765631479493087,"train/train/tensor_grad_model_layers_47_mlp_waleed_W_u_weight/norm":0.013321438459312702,"train/train/tensor_act_model_layers_67_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_mlp_waleed_W_g_weight/norm":0.018097001022508736,"train/train/tensor_act_model_layers_88_self_attn_k_proj/max_abs":6.75,"train/train/tensor_act_model_layers_45_self_attn/mean":-0.00096893310546875,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_mlp_waleed_W_u/norm":5926.289870866271,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/std":0.03515625,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/norm":0.020453062087997537,"train/train/tensor_act_model_layers_33_mlp_waleed_W_u/max_abs":2.03125,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_v_proj/std":0.3242217378261428,"train/train/tensor_act_model_layers_60/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/max_abs":0.002288818359375,"train/train/tensor_param_model_layers_74_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_mlp_waleed_W_u_weight/norm":0.022738606298417334,"train/train/tensor_act_model_layers_46_self_attn_q_proj/max_abs":6.09375,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/std":0.05615234375,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_37_input_layernorm/norm":5792.610595706896,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/norm":0.01563203176582012,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_mlp_waleed_W_u_weight/std":5.527314890221489e-05,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/std":0.0283203125,"train/train/tensor_act_model_layers_58_post_attention_layernorm/norm":5792.609375000854,"train/train/layer_model_layers_62/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_mlp_waleed_W_u_weight/std":3.9221901180823144e-05,"train/train/tensor_param_model_layers_61_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/norm":0.015352070826687866,"train/train/tensor_act_model_layers_31_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn/mean":0.00012445449829101562,"train/train/tensor_act_model_layers_49_self_attn_k_proj/mean":0.04095458984375,"train/train/tensor_act_model_layers_12_post_attention_layernorm/max_abs":5.15625,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_69_post_attention_layernorm/mean":0.00460052490234375,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/mean":-8.361530490219593e-08,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/max_abs":0.1630859375,"train/train/tensor_param_model_layers_13_mlp_waleed_W_u_weight/norm":4.1875,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_22_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp/mean":-0.0056915283203125,"train/train/tensor_act_model_layers_63_self_attn_k_proj/norm":5819.882389579524,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/mean":8.630752563476562e-05,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/max_abs":0.0003528594970703125,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/mean":4.540197551250458e-07,"train/train/layer_model_layers_52/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/norm":4.625,"train/train/tensor_act_model_layers_86/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/norm":0.005044962733299335,"train/train/tensor_act_model_layers_11_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp_waleed/frac_near_user_limit":0,"train/train/layer_model_layers_19/act/mean":-0.001378089189529419,"train/train/tensor_act_model_layers_28_self_attn/norm":1165.626200868957,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/mean":3.2014213502407074e-08,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/std":0.00012208762103008215,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/max_abs":0.00014019012451171875,"train/train/tensor_act_model_layers_22_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/std":4.285722692927383e-05,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/mean":0.0002288818359375,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_32/act/max_abs":26.375,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/max_abs":0.0004253387451171875,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/norm":4.28125,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/std":4.488642170247605e-05,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/mean":-0.00011444091796875,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/mean":-1.4650140656158328e-07,"train/train/tensor_param_model_layers_7_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/norm":0.01399436444717806,"train/train/tensor_param_model_layers_27_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/max_abs":0.000278472900390625,"train/train/tensor_param_model_layers_93_mlp_waleed_W_g_weight/std":0.0673828125,"train/train/tensor_param_model_layers_86_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/mean":4.3818727135658264e-07,"train/train/tensor_act_model_layers_45_mlp_waleed/std":0.10974141651614726,"train/train/tensor_act_model_layers_87_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/max_abs":0.00122833251953125,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_19_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_24_mlp_waleed_W_g/max_abs":2.234375,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/norm":0.014267645019123774,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_input_layernorm/std":1.0000003636669326,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/norm":0.007029036994872206,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/std":0.00013812098853736393,"train/train/tensor_act_model_layers_3_post_attention_layernorm/std":1.0000000678410266,"train/train/layer_model_layers_51/act/std":0.8634386341262341,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/mean":0.0001010894775390625,"train/train/layer__model_layers_3/param/norm":18.922932980331694,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/max_abs":0.1357421875,"train/train/tensor_act_model_layers_53/std":2.6094002675486085,"train/train/tensor_act_model_layers_68_self_attn/std":0.14062502625165263,"train/train/tensor_act_model_layers_21_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20/norm":16239.667034613347,"train/train/tensor_param_model_layers_23_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/max_abs":0.0029449462890625,"train/train/tensor_act_model_layers_15_mlp_waleed_W_u/mean":0.0006875991821289062,"train/train/tensor_act_model_layers_35_self_attn_v_proj/max_abs":2.078125,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/std":0.00010003499926538882,"train/train/tensor_act_model_layers_14_mlp_waleed/mean":-0.0003018379211425781,"train/train/tensor_act_model_layers_53_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_q_proj/std":0.9384781344238835,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/std":0.00011001821473742317,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/layer_model_layers_5/act/max_abs":26.875,"train/train/tensor_grad_model_layers_29_mlp_waleed_W_g_weight/mean":1.3198587112128735e-08,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/max_abs":0.2294921875,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_2_self_attn_q_proj/mean":0.028045654296875,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_26/grad/max_abs":0.0014801025390625,"train/train/tensor_act_model_layers_34_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_38/norm":15293.595141798658,"train/train/tensor_act_model_layers_20_input_layernorm/norm":5792.6085205119525,"train/train/tensor_act_model_layers_32_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/norm":4.75,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/norm":0.04314283306325062,"train/train/tensor_act_model_layers_78/max_abs":26,"train/train/tensor_act_model_layers_73_mlp_down_proj/std":0.15405332950816295,"train/train/tensor_act_model_layers_62_mlp_down_proj/std":0.09460481517705308,"train/train/tensor_act_model_layers_76_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/max_abs":0.154296875,"train/train/tensor_act_model_layers_56_self_attn_v_proj/norm":2229.913182075164,"train/train/tensor_grad_model_layers_46_mlp_waleed_W_g_weight/norm":0.01521377025925727,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_mlp_waleed_W_g/mean":-0.0055694580078125,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/std":0.05029296875,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/std":0.03564453125,"train/train/tensor_act_model_layers_55_mlp_waleed/max_abs":4.46875,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/norm":2.90625,"train/train/tensor_act_model_layers_77_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_mlp_waleed_W_g_weight/mean":-5.602836608886719e-05,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn_v_proj/norm":1972.1109834800372,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/std":0.03564453125,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/max_abs":0.000507354736328125,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/max_abs":0.1767578125,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/mean":0.0002727508544921875,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/max_abs":0.000583648681640625,"train/train/tensor_act_model_layers_29_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/mean":-4.76837158203125e-07,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/mean":-2.6404450181871653e-08,"train/train/tensor_act_model_layers_35_self_attn/mean":-0.00029662251472473145,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/std":0.05029296875,"train/train/tensor_act_model_layers_60_self_attn_v_proj/max_abs":2.1875,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/norm":3.71875,"train/train/tensor_act_model_layers_30_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_o_proj/norm":160.16051710661873,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/mean":-2.3126602172851562e-05,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/std":0.00015520462239268593,"train/train/tensor_act_model_layers_20_mlp_waleed_W_g/std":0.24877966713714,"train/train/tensor_param_model_layers_49_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/mean":-1.9073486328125e-05,"train/train/layer_model_layers_2/grad/std":0.00020415478216659867,"train/train/tensor_param_model_layers_21_mlp_waleed_W_g_weight/max_abs":0.1162109375,"train/train/tensor_act_model_layers_75_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn/max_abs":1.4765625,"train/train/tensor_act_model_layers_52_self_attn_o_proj/norm":323.94372028663236,"train/train/layer__model_layers_17/param/max_abs":1,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_79/act/norm":22833.099336145006,"train/train/tensor_act_model_layers_66_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_waleed_W_u_weight/norm":6.40625,"train/train/tensor_act_model_layers_86_mlp/max_abs":3.953125,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/norm":5.96875,"train/train/layer__model_layers_90/param/mean":0.0017227196656225624,"train/train/tensor_act_model_layers_89_self_attn_k_proj/mean":-0.022705078125,"train/train/tensor_grad_model_layers_92_mlp_waleed_W_g_weight/std":0.00012231574762064224,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_input_layernorm/max_abs":5.15625,"train/train/tensor_act_model_layers_4_mlp/mean":0.01007080078125,"train/train/tensor_param_model_layers_50_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3/norm":19557.356362813545,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/max_abs":0.0004520416259765625,"train/train/tensor_act_model_layers_31_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_mlp_waleed_W_u_weight/norm":0.029270605389577445,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp/norm":1422.8903211425861,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/max_abs":0.0008697509765625,"train/train/layer_model_layers_28/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_waleed/mean":0.01861572265625,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/mean":-5.7220458984375e-05,"train/train/tensor_grad_model_layers_12_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/max_abs":8.20159912109375e-05,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/mean":0.00547027587890625,"train/train/tensor_act_model_layers_87_mlp_waleed_W_u/norm":5867.315104881721,"train/train/tensor_act_model_layers_16_mlp_down_proj/max_abs":0.498046875,"train/train/tensor_grad_model_layers_61_mlp_waleed_W_u_weight/max_abs":0.0012359619140625,"train/train/tensor_act_model_layers_89_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/max_abs":0.0010223388671875,"train/train/tensor_param_model_layers_87_mlp_waleed_W_g_weight/mean":-0.000194549560546875,"train/train/tensor_param_model_layers_78_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_72/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_48_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/std":0.024169921875,"train/train/tensor_act_model_layers_66_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/layer_model_layers_77/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_57/grad/max_abs":0.00121307373046875,"train/train/tensor_grad_model_layers_37_mlp_waleed_W_u_weight/std":4.0117970170251877e-05,"train/train/tensor_act_model_layers_72_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/std":0.00077056884765625,"train/train/tensor_act_model_layers_88_input_layernorm/max_abs":6.28125,"train/train/tensor_act_model_layers_11_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_22/grad/mean":-4.571417198398742e-08,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_v_proj/max_abs":1.7734375,"train/train/layer__model_layers_82/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/max_abs":0.000431060791015625,"train/train/tensor_grad_model_layers_24_mlp_waleed_W_u_weight/max_abs":0.000701904296875,"train/train/tensor_act_model_layers_64_post_attention_layernorm/mean":0.0041656494140625,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/norm":2.78125,"train/train/tensor_act_model_layers_46_post_attention_layernorm/std":1.0000015169705283,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/max_abs":0.1796875,"train/train/tensor_act_model_layers_36_self_attn_o_proj/mean":-0.00415802001953125,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/mean":-4.425644874572754e-06,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/max_abs":0.1318359375,"train/train/tensor_act_model_layers_50_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_mlp_waleed_W_g_weight/norm":0.03375922853960159,"train/train/tensor_act_model_layers_40_mlp/max_abs":0.5703125,"train/train/tensor_param_model_layers_50_mlp_waleed_W_u_weight/max_abs":0.1376953125,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/max_abs":0.000514984130859375,"train/train/tensor_grad_model_layers_26_mlp_waleed_W_g_weight/mean":4.380854079499841e-08,"train/train/tensor_act_model_layers_4_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn_q_proj/norm":9216.61793508373,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/norm":5.53125,"train/train/layer__model_layers_47/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/std":1.9251194161665406e-05,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/mean":-3.14321368932724e-08,"train/train/tensor_act_model_layers_45_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/mean":-0.00019550323486328125,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_q_proj/norm":6137.864096025527,"train/train/layer_model_layers_10/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/max_abs":0.00010538101196289062,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/std":4.717843046195099e-05,"train/train/tensor_act_model_layers_91_post_attention_layernorm/norm":5792.609130861623,"train/train/tensor_act_model_layers_32_mlp_down_proj/norm":417.60883952066683,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_64_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/mean":9.229406714439392e-07,"train/train/tensor_act_model_layers_38_self_attn_q_proj/max_abs":10.5625,"train/train/tensor_act_model_layers_36_mlp/mean":0.0010528564453125,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_waleed_W_u_weight/mean":8.249282836914062e-05,"train/train/tensor_param_model_layers_35_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_19/param/std":0.04731247549679916,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/mean":9.729410521686077e-08,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/norm":0.001236303305866438,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/std":0.0001561071363618151,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/norm":5.03125,"train/train/tensor_act_model_layers_77_self_attn_o_proj/mean":0.0089569091796875,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/max_abs":0.00104522705078125,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/mean":6.193295121192932e-07,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/norm":5.625,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_41/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn/max_abs":2.46875,"train/train/tensor_act_model_layers_75_post_attention_layernorm/max_abs":5.4375,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/norm":0.007409143822119224,"train/train/tensor_param_model_layers_9_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/std":2.8827521160162528e-05,"train/train/tensor_param_model_layers_71_mlp_waleed_W_u_weight/mean":-5.9604644775390625e-05,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_mlp_waleed_W_u_weight/mean":-9.860377758741379e-08,"train/train/layer_model_layers_87/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/mean":1.1026859283447266e-05,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/max_abs":0.000667572021484375,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/norm":0.0014501570526367852,"train/train/tensor_act_model_layers_10_input_layernorm/norm":5792.609863288453,"train/train/layer_model_layers_58/grad/max_abs":0.001495361328125,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/std":3.0370433670598238e-05,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_waleed_W_g_weight/std":0.0272216796875,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_post_attention_layernorm/max_abs":5.21875,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/std":0.023681640625,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/max_abs":0.00089263916015625,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92/frac_near_dtype_limit":0,"train/train/layer_model_layers_84/grad/frac_near_dtype_limit":0,"train/train/layer__model_layers_44/param/norm":19.757526750108426,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/max_abs":0.14453125,"train/train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_input_layernorm/mean":0.002445220947265625,"train/train/tensor_act_model_layers_30_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_post_attention_layernorm/std":1.0000000284926496,"train/train/tensor_grad_model_layers_81_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_q_proj/norm":6806.20811021515,"train/train/tensor_param_model_layers_67_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/mean":-0.00010824203491210938,"train/train/tensor_act_model_layers_28_self_attn_v_proj/mean":0.00457763671875,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_23/act/std":0.9343808550188809,"train/train/tensor_param_model_layers_61_mlp_waleed_W_g_weight/max_abs":0.1572265625,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/std":8.550390252967919e-05,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/max_abs":0.279296875,"train/train/tensor_act_model_layers_13_post_attention_layernorm/std":1.0000000163854565,"train/train/tensor_param_model_layers_1_mlp_waleed_W_u_weight/std":0.029541015625,"train/train/tensor_act_model_layers_82_mlp/std":0.355468826962033,"train/train/tensor_act_model_layers_82_mlp_waleed_W_u/max_abs":3.734375,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/norm":0.030370311175012363,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/mean":-2.278946340084076e-06,"train/train/tensor_param_model_layers_14_mlp_waleed_W_g_weight/std":0.02294921875,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/max_abs":0.162109375,"train/train/tensor_act_model_layers_34_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_down_proj/max_abs":0.66796875,"train/train/tensor_param_model_layers_47_mlp_waleed_W_g_weight/max_abs":0.1611328125,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/mean":-1.1008232831954956e-06,"train/train/tensor_act_model_layers_69_self_attn_v_proj/norm":2683.265686102265,"train/train/tensor_param_model_layers_83_mlp_waleed_W_g_weight/max_abs":0.2431640625,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/std":0.03466796875,"train/train/layer_model_layers_70/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_46_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_37_mlp_waleed_W_g_weight/norm":4.625,"train/train/tensor_act_model_layers_9_self_attn_o_proj/norm":555.7039763433191,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/mean":-3.1152740120887756e-07,"train/train/tensor_act_model_layers_58/norm":15014.968860366944,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/mean":-3.9581209421157837e-07,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_65/param/std":0.052258016003176824,"train/train/tensor_grad_model_layers_8_mlp_waleed_W_u_weight/mean":-2.058914105873555e-08,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/mean":-0.00013637542724609375,"train/train/tensor_act_model_layers_1_self_attn_o_proj/max_abs":1.9296875,"train/train/tensor_act_model_layers_32_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_v_proj/norm":2374.4187530972486,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/max_abs":0.10986328125,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78/norm":17153.969878635682,"train/train/layer_model_layers_79/act/max_abs":26.625,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_9_mlp_waleed_W_g_weight/std":6.354785223693279e-05,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/std":3.959851067254733e-05,"train/train/tensor_act_model_layers_50_input_layernorm/max_abs":5.1875,"train/train/tensor_act_model_layers_34_self_attn_o_proj/mean":0.0005230903625488281,"train/train/tensor_param_model_layers_77_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/std":0.00015166604142807178,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/norm":4.21875,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/std":9.585814244294373e-05,"train/train/layer_model_layers_4/grad/mean":2.1128236736038733e-07,"train/train/tensor_act_model_layers_61_post_attention_layernorm/max_abs":4.90625,"train/train/tensor_act_model_layers_83_mlp_waleed_W_u/norm":4767.609790715901,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/std":3.770746307774113e-05,"train/train/tensor_grad_model_layers_56_mlp_waleed_W_g_weight/norm":0.015875946932808013,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/norm":0.024539624625213165,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/mean":-9.32253897190094e-07,"train/train/tensor_act_model_layers_7_self_attn_v_proj/mean":0.011627197265625,"train/train/tensor_act_model_layers_56_input_layernorm/std":1.000001054267854,"train/train/tensor_act_model_layers_48_mlp_down_proj/mean":-0.00313568115234375,"train/train/tensor_grad_model_layers_54_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer__model_layers_47/param/std":0.048161370776880846,"train/train/layer_model_layers_88/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/norm":0.02002083378718042,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/std":4.69887952695462e-05,"train/train/tensor_grad_model_layers_61_mlp_waleed_W_u_weight/mean":-2.123415470123291e-07,"train/train/tensor_param_model_layers_51_mlp_waleed_W_u_weight/max_abs":0.13671875,"train/train/tensor_grad_model_layers_0_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_waleed_W_g_weight/mean":0.0001811981201171875,"train/train/tensor_act_model_layers_39_mlp_down_proj/norm":366.7478613308848,"train/train/tensor_act_model_layers_21/mean":-0.004940032958984375,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/mean":-2.275337465107441e-07,"train/train/tensor_act_model_layers_3_self_attn_v_proj/norm":1905.288513153094,"train/train/layer_model_layers_33/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_mlp_waleed_W_u_weight/std":0.04150390625,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/mean":3.247987478971481e-08,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_post_attention_layernorm/mean":-0.00262451171875,"train/train/tensor_act_model_layers_17_mlp/norm":409.8246234998011,"train/train/tensor_grad_model_layers_82_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_mlp_waleed_W_u_weight/std":0.0478515625,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_embed_tokens_weight/mean":0.0002040863037109375,"train/train/tensor_act_model_layers_71_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/norm":0.00028043538803493594,"train/train/layer_model_layers_7/act/std":1.0152180997127231,"train/train/tensor_act_model_layers_45_post_attention_layernorm/norm":5792.611083988393,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/mean":-3.427267074584961e-07,"train/train/tensor_param_model_layers_58_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_34_mlp_waleed_W_u/std":0.2714843915008245,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/norm":3.09375,"train/train/tensor_param_model_layers_27_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/norm":0.017404118885135732,"train/train/tensor_grad_model_layers_79_mlp_waleed_W_g_weight/norm":0.02591105928864549,"train/train/tensor_act_model_layers_2_mlp/max_abs":3.890625,"train/train/tensor_act_model_layers_23_mlp_waleed_W_g/std":0.23413126473595589,"train/train/tensor_act_model_layers_21_self_attn_o_proj/std":0.24487368773690996,"train/train/layer_model_layers_12/grad/mean":-4.925984488262587e-08,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_mlp_waleed_W_g_weight/mean":-1.6007106751203537e-07,"train/train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_down_proj/norm":1210.9663292434855,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/std":0.029541015625,"train/train/tensor_act_model_layers_78_mlp_waleed_W_g/max_abs":4.21875,"train/train/tensor_act_model_layers_21_mlp_waleed_W_g/norm":2332.8127439057835,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_mlp_waleed_W_g_weight/mean":-0.00022792816162109375,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/norm":0.005083375747072482,"train/train/tensor_act_model_layers_93_mlp/norm":16964.358780560604,"train/train/tensor_act_model_layers_19_mlp_waleed/frac_near_dtype_limit":0,"train/train/layer_model_layers_53/grad/mean":-1.1093599224890265e-08,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/mean":-5.8033037930727005e-08,"train/train/tensor_act_model_layers_46/max_abs":24.5,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_76_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_o_proj/norm":997.9807777449652,"train/train/tensor_act_model_layers_67_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/std":0.0274658203125,"train/train/tensor_grad_model_layers_27_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/std":0.0001977322486041798,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/max_abs":0.1884765625,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/mean":-4.3120235204696655e-07,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_rotary_emb/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp_waleed_W_g/norm":7749.383682398301,"train/train/tensor_act_model_layers_14_mlp_waleed_W_g/max_abs":3.5,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/max_abs":0.0003604888916015625,"train/train/tensor_act_model_layers_40_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/norm":0.0006199587268443217,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_1/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp_waleed/mean":-0.0014553070068359375,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/std":0.05126953125,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/mean":1.5404075384140015e-06,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/norm":0.0032758989013662084,"train/train/layer__model_layers_91/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_mlp_waleed_W_u_weight/std":4.167925980959578e-05,"train/train/tensor_act_model_layers_48_self_attn_o_proj/max_abs":4.78125,"train/train/tensor_param_model_layers_7_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_33_mlp_waleed_W_u_weight/std":0.024658203125,"train/train/layer_model_layers_84/grad/max_abs":0.002288818359375,"train/train/tensor_param_model_layers_33_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/mean":-4.7206878662109375e-05,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/max_abs":0.000164031982421875,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/max_abs":0.1162109375,"train/train/tensor_act_model_layers_14_mlp_waleed_W_g/mean":0.0004162788391113281,"train/train/tensor_act_model_layers_52_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_52/act/std":0.8251624765170106,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/max_abs":0.1259765625,"train/train/tensor_act_model_layers_93_self_attn/norm":3502.8440548790777,"train/train/tensor_act_model_layers_13_self_attn_v_proj/std":0.36914067041305987,"train/train/layer_model_layers_35/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/std":0.02490234375,"train/train/tensor_act_model_layers_55_mlp_waleed_W_u/mean":-0.0013904571533203125,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/mean":-0.00010061264038085938,"train/train/tensor_act_model_layers_9_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/std":0.0240478515625,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_87_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/max_abs":0.000774383544921875,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_mlp_waleed_W_g_weight/mean":-3.5762786865234375e-05,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/max_abs":0.140625,"train/train/tensor_grad_model_layers_25_mlp_waleed_W_u_weight/norm":0.014436137035983886,"train/train/tensor_param_model_layers_53_mlp_waleed_W_g_weight/std":0.0281982421875,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/mean":6.99758529663086e-05,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/mean":3.159046173095703e-06,"train/train/tensor_act_model_layers_63_mlp_waleed_W_u/mean":0.01641845703125,"train/train/tensor_act_model_layers_22_input_layernorm/max_abs":5.21875,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/norm":5.1875,"train/train/tensor_act_model_layers_6_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_mlp_down_proj/std":0.10388207598753144,"train/train/tensor_act_model_layers_40_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp_waleed_W_u/std":1.2109376153638232,"train/train/tensor_grad_model_layers_51_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_69/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp_waleed/frac_near_user_limit":0,"train/train/layer__model_layers_44/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_mlp_waleed_W_u_weight/std":4.029626997559113e-05,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/max_abs":0.2490234375,"train/train/tensor_act_model_layers_5_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_mlp_waleed_W_g_weight/max_abs":0.00078582763671875,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/norm":0.004948766639550938,"train/train/tensor_grad_model_layers_87_mlp_waleed_W_g_weight/std":9.098956576441727e-05,"train/train/tensor_act_model_layers_48_self_attn_q_proj/mean":0.0201416015625,"train/train/tensor_act_model_layers_40_mlp_waleed_W_g/norm":2525.4883505541525,"train/train/tensor_act_model_layers_79_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/max_abs":0.0005035400390625,"train/train/tensor_act_model_layers_3_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_waleed_W_u/std":0.8066438616912359,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/max_abs":0.00090789794921875,"train/train/tensor_act_model_layers_4_self_attn_q_proj/norm":8116.720058647966,"train/train/layer_model_layers_21/act/mean":0.0005766749382019043,"train/train/tensor_act_model_layers_49_self_attn_k_proj/norm":4224.286723711923,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/mean":7.175840437412262e-07,"train/train/layer__model_layers_67/param/std":0.05429634125478296,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/std":5.724968684920162e-05,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_down_proj/mean":0.0112762451171875,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/std":0.04248046875,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/mean":-0.000518798828125,"train/train/tensor_grad_model_layers_53_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_mlp_waleed_W_g/max_abs":2.390625,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_4_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_mlp_waleed_W_g_weight/norm":0.01852565300674718,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/norm":4.75,"train/train/tensor_act_model_layers_90_mlp/max_abs":7.125,"train/train/tensor_act_model_layers_19_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_q_proj/norm":5726.918857637197,"train/train/layer_model_layers_55/grad/max_abs":0.0011444091796875,"train/train/layer__model_layers_36/param/norm":20.087759118583886,"train/train/tensor_act_model_layers_61_mlp/std":0.09863283671229484,"train/train/tensor_act_model_layers_63_self_attn_k_proj/mean":0.0440673828125,"train/train/tensor_act_model_layers_11_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_mlp_waleed/mean":0.0003857612609863281,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/max_abs":0.00128173828125,"train/train/tensor_act_model_layers_18_self_attn_k_proj/max_abs":5.5,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp/mean":0.003726959228515625,"train/train/tensor_act_model_layers_68_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp_waleed/norm":1967.2365339867836,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/norm":0.016085629868310074,"train/train/tensor_grad_model_layers_85_mlp_waleed_W_u_weight/std":8.359255651878029e-05,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_mlp_waleed_W_u_weight/max_abs":0.177734375,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_input_layernorm/norm":5792.603515629229,"train/train/tensor_act_model_layers_57_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15/max_abs":26.25,"train/train/layer__model_layers_52/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_mlp_waleed_W_g_weight/mean":3.5315752029418945e-06,"train/train/tensor_act_model_layers_39_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/max_abs":0.0010986328125,"train/train/tensor_param_model_layers_45_mlp_waleed_W_u_weight/mean":-1.2993812561035156e-05,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/mean":-3.9577484130859375e-05,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/std":0.036865234375,"train/train/tensor_act_model_layers_66_mlp_down_proj/mean":0.000576019287109375,"train/train/tensor_act_model_layers_78_self_attn_v_proj/std":0.46679704637890923,"train/train/tensor_act_model_layers_79_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_v_proj/mean":0.00626373291015625,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/mean":-0.00010967254638671875,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/max_abs":0.1923828125,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_75/max_abs":25.75,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_93_mlp_waleed/norm":12168.88407003216,"train/train/tensor_act_model_layers_62_mlp_waleed_W_u/std":0.38134861165936246,"train/train/tensor_act_model_layers_26_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_43/grad/norm":0.033429999206343805,"train/train/tensor_act_model_layers_1_mlp_down_proj/mean":0.0003580152988433838,"train/train/tensor_act_model_layers_4_self_attn/norm":491.34969060766144,"train/train/layer_model_layers_20/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_mlp_waleed_W_u_weight/norm":0.030313711142186117,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_mlp_waleed_W_g_weight/max_abs":0.111328125,"train/train/tensor_act_model_layers_88/max_abs":33,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_waleed_W_u_weight/mean":0.00021839141845703125,"train/train/tensor_act_model_layers_16_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_mlp_waleed_W_u_weight/max_abs":0.12109375,"train/train/layer_model_layers_37/grad/max_abs":0.000942230224609375,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/mean":0.0002155303955078125,"train/train/tensor_act_model_layers_4/mean":-0.00952911376953125,"train/train/tensor_grad_model_layers_88_mlp_waleed_W_g_weight/mean":-8.997449185699224e-08,"train/train/tensor_grad_model_layers_93_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/std":0.00010252278794329426,"train/train/tensor_grad_model_layers_7_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_mlp_waleed_W_g_weight/max_abs":0.1025390625,"train/train/tensor_param_model_layers_50_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_53_mlp_down_proj/norm":475.5803465210237,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_post_attention_layernorm/norm":5792.610961917966,"train/train/tensor_act_model_layers_79_self_attn_o_proj/norm":1744.0613024313504,"train/train/tensor_act_model_layers_76/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp_waleed_W_u/mean":0.002178192138671875,"train/train/tensor_act_model_layers_16_input_layernorm/max_abs":5.09375,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/std":0.0002016178206688167,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_waleed_W_u_weight/max_abs":0.002197265625,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_19_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_28_mlp_waleed_W_u_weight/norm":4.40625,"train/train/tensor_act_model_layers_13_self_attn_v_proj/norm":2139.5982702227407,"train/train/tensor_act_model_layers_32_self_attn_v_proj/mean":-0.00010669231414794922,"train/train/tensor_grad_model_layers_69_mlp_waleed_W_g_weight/max_abs":0.0006103515625,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_waleed_W_u_weight/norm":4.21875,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/std":0.00011863403036036269,"train/train/tensor_grad_model_layers_84_mlp_waleed_W_g_weight/mean":3.2922253012657166e-07,"train/train/tensor_act_model_layers_21_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/mean":-1.4603137969970703e-05,"train/train/tensor_act_model_layers_51_self_attn_k_proj/max_abs":5.875,"train/train/layer_model_layers_54/act/std":0.8258796349216706,"train/train/tensor_param_model_layers_86_mlp_waleed_W_g_weight/std":0.045654296875,"train/train/layer_model_layers_63/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_71_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_36_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_30_mlp_waleed_W_u/max_abs":2,"train/train/tensor_param_model_layers_69_mlp_waleed_W_u_weight/std":0.033203125,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/max_abs":0.0002956390380859375,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/mean":4.093162715435028e-07,"train/train/tensor_act_model_layers_37/mean":0.006160736083984375,"train/train/tensor_act_model_layers_17_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_o_proj/std":0.17212814355224587,"train/train/tensor_act_model_layers_4_mlp_down_proj/mean":0.01007080078125,"train/train/tensor_act_model_layers_68_mlp_waleed/std":0.16674858628307243,"train/train/tensor_param_model_layers_6_mlp_waleed_W_u_weight/mean":4.649162292480469e-05,"train/train/tensor_act_model_layers_81_self_attn_v_proj/max_abs":4.03125,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/max_abs":0.00014019012451171875,"train/train/tensor_act_model_layers_14_self_attn_q_proj/std":0.9511738822179525,"train/train/tensor_act_model_layers_28/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/max_abs":0.00023651123046875,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_self_attn_q_proj/std":1.2363328059802396,"train/train/tensor_grad_model_layers_41_mlp_waleed_W_u_weight/std":4.005610949218096e-05,"train/train/tensor_grad_model_layers_6_mlp_waleed_W_u_weight/std":0.00012688349946662995,"train/train/tensor_act_model_layers_68_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_waleed_W_u_weight/std":0.044189453125,"train/train/layer_model_layers_3/grad/mean":-6.380967153657617e-08,"train/train/tensor_act_model_layers_42_self_attn_o_proj/max_abs":1.015625,"train/train/tensor_act_model_layers_27_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_46/act/norm":19243.435362696648,"train/train/tensor_act_model_layers_81/max_abs":25.75,"train/train/tensor_act_model_layers_86_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_84/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/max_abs":0.000885009765625,"train/train/tensor_act_model_layers_70/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_post_attention_layernorm/std":1.0000001646112515,"train/train/tensor_act_model_layers_29_self_attn_q_proj/max_abs":6.25,"train/train/tensor_act_model_layers_18_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/mean":-1.8477439880371094e-05,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/mean":2.514570951461792e-07,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/norm":0.0029555116393445847,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/std":0.00011706919195333846,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/max_abs":0.0004863739013671875,"train/train/tensor_act_model_layers_78_self_attn_v_proj/norm":2701.1541057412473,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_34/grad/mean":-8.934267539408947e-08,"train/train/tensor_grad_model_layers_23_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_waleed/norm":1474.6269129325663,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/mean":1.3422686606645584e-07,"train/train/tensor_act_model_layers_76_mlp_down_proj/norm":1085.3489554907305,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/std":0.053955078125,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/std":0.00013623287259918447,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/std":4.028255235531686e-05,"train/train/global/param/max_abs":1,"train/train/tensor_act_model_layers_16_mlp/std":0.058166600886850856,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_66_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_waleed/mean":0.000244140625,"train/train/layer_model_layers_85/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/std":0.052490234375,"train/train/tensor_param_model_layers_69_mlp_waleed_W_g_weight/std":0.033203125,"train/train/tensor_act_model_layers_91_self_attn_k_proj/norm":7266.332463643814,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/mean":2.6411726139485836e-08,"train/train/tensor_act_model_layers_18_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_mlp_waleed_W_g_weight/norm":5.6875,"train/train/tensor_param_model_layers_81_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_55/act/std":0.8316871034117931,"train/train/tensor_act_model_layers_43_mlp/mean":0.00174713134765625,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/std":0.03759765625,"train/train/layer__model_layers_39/param/max_abs":1,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/norm":4.625,"train/train/tensor_param_model_layers_46_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_45_self_attn_k_proj/mean":-0.03436279296875,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/max_abs":0.00020885467529296875,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_mlp_waleed_W_u_weight/std":0.045654296875,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/norm":0.0012047500059317443,"train/train/tensor_act_model_layers_47_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_norm_weight/mean":-0.0043792724609375,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/max_abs":0.2294921875,"train/train/tensor_act_model_layers_38_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_45/act/mean":-0.0023550987243652344,"train/train/tensor_act_model_layers_78_input_layernorm/mean":0.007232666015625,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/norm":0.015641352917617576,"train/train/tensor_act_model_layers_48_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_waleed/mean":0.0026092529296875,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55/norm":15031.33655924907,"train/train/layer_model_layers_91/act/std":1.4140035294733972,"train/train/tensor_act_model_layers_58_self_attn/std":0.19751162106969405,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/std":0.02490234375,"train/train/tensor_param_model_layers_89_mlp_waleed_W_u_weight/max_abs":0.251953125,"train/train/tensor_act_model_layers_46_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17/mean":-0.009063720703125,"train/train/tensor_act_model_layers_22_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_post_attention_layernorm/norm":5792.612548830743,"train/train/tensor_act_model_layers_23_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_waleed_W_g_weight/std":0.044189453125,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/norm":5.4375,"train/train/tensor_grad_model_layers_76_mlp_waleed_W_u_weight/std":6.795481095808407e-05,"train/train/tensor_act_model_layers_68_mlp_waleed_W_u/norm":3363.491706825414,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp_down_proj/max_abs":0.49609375,"train/train/tensor_act_model_layers_20_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_v_proj/max_abs":2.515625,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_q_proj/std":1.2734377735231257,"train/train/tensor_act_model_layers_43_post_attention_layernorm/mean":0.003467559814453125,"train/train/tensor_act_model_layers_73_self_attn/std":0.3291031018482552,"train/train/tensor_act_model_layers_45_self_attn_k_proj/std":0.9970718424875923,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_embed_tokens_weight/max_abs":0.036376953125,"train/train/tensor_act_model_layers_22_mlp/std":0.07409698298810914,"train/train/tensor_act_model_layers_22_mlp_waleed/max_abs":2.484375,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/max_abs":0.1162109375,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/max_abs":0.00067138671875,"train/train/tensor_act_model_layers_66_self_attn/std":0.22583156804289212,"train/train/tensor_param_model_layers_61_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_64/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp/max_abs":1.5234375,"train/train/tensor_grad_model_layers_90_mlp_waleed_W_u_weight/norm":0.03675958513613141,"train/train/tensor_act_model_layers_90_mlp_waleed_W_u/max_abs":5.875,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/max_abs":0.00014591217041015625,"train/train/tensor_grad_model_layers_38_mlp_waleed_W_g_weight/max_abs":0.00084686279296875,"train/train/tensor_act_model_layers_86_mlp_waleed/max_abs":7.9375,"train/train/tensor_act_model_layers_89_mlp_waleed_W_g/max_abs":4.40625,"train/train/tensor_grad_model_layers_62_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_mlp_waleed_W_u_weight/norm":0.024081521257010834,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/std":3.262208884323179e-05,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_58/param/mean":0.0015781301418072349,"train/train/tensor_act_model_layers_70_mlp/max_abs":0.91015625,"train/train/tensor_act_model_layers_80_self_attn_v_proj/std":0.6445371902076489,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/max_abs":0.1494140625,"train/train/tensor_act_model_layers_1_self_attn/std":0.13110787834601914,"train/train/tensor_param_model_layers_91_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/norm":0.027744691568696175,"train/train/tensor_grad_model_layers_14_mlp_waleed_W_g_weight/max_abs":0.00616455078125,"train/train/tensor_act_model_layers_83_mlp_down_proj/max_abs":2.796875,"train/train/tensor_grad_model_layers_0_mlp_waleed_W_g_weight/mean":5.7443976402282715e-06,"train/train/tensor_act_model_layers_24_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/norm":4.25,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_input_layernorm_weight/mean":1,"train/train/layer_model_layers_50/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/std":0.04443359375,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/std":3.3971981506500726e-05,"train/train/tensor_act_model_layers_39_self_attn_o_proj/std":0.14868266414880968,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/mean":4.265457391738892e-06,"train/train/tensor_act_model_layers_20_self_attn_v_proj/norm":2402.136588275805,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/mean":2.4971086531877518e-08,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/max_abs":0.00015926361083984375,"train/train/tensor_act_model_layers_15_mlp_waleed/max_abs":2.609375,"train/train/tensor_act_model_layers_51_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_mlp_waleed_W_u/norm":3784.3355372973024,"train/train/tensor_act_model_layers_21_self_attn_o_proj/mean":-0.0009427070617675781,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp_waleed_W_u/std":0.23828128358868064,"train/train/tensor_act_model_layers_81_mlp_waleed_W_g/mean":0.01763916015625,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/norm":3.671875,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/mean":2.907589077949524e-06,"train/train/tensor_act_model_layers_12_mlp_down_proj/max_abs":0.5078125,"train/train/tensor_act_model_layers_62_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62/max_abs":24.5,"train/train/tensor_act_model_layers_87_self_attn_q_proj/max_abs":6.375,"train/train/tensor_act_model_layers_46_mlp_waleed_W_g/max_abs":2.703125,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/mean":-4.623143468052149e-08,"train/train/tensor_act_model_layers_38_input_layernorm/mean":0.0030994415283203125,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/mean":9.518043952994049e-08,"train/train/tensor_act_model_layers_23/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_o_proj/std":0.10779024619608649,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/max_abs":0.1337890625,"train/train/tensor_grad_model_layers_64_mlp_waleed_W_u_weight/mean":4.423782229423523e-08,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_waleed/max_abs":2.984375,"train/train/layer_model_layers_61/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_k_proj/mean":0.0149383544921875,"train/train/tensor_act_model_layers_27_self_attn_k_proj/norm":6053.213539691632,"train/train/tensor_act_model_layers_86_mlp_waleed/mean":0.00131988525390625,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/max_abs":0.12353515625,"train/train/tensor_act_model_layers_44_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/norm":7.15625,"train/train/tensor_act_model_layers_59/std":2.5937752799211853,"train/train/tensor_act_model_layers_38_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/norm":0.018440497526482316,"train/train/tensor_param_model_layers_55_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_18_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_post_attention_layernorm/norm":5792.605102544438,"train/train/tensor_act_model_layers_90_post_attention_layernorm/mean":-0.0027713775634765625,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/max_abs":0.0004425048828125,"train/train/tensor_act_model_layers_14_mlp_down_proj/std":0.06921419466199902,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/norm":4.9375,"train/train/tensor_act_model_layers_81_input_layernorm/mean":0.011383056640625,"train/train/tensor_act_model_layers_1_self_attn_o_proj/std":0.13110787834601914,"train/train/tensor_act_model_layers_50_input_layernorm/std":1.000001178847684,"train/train/tensor_act_model_layers_62_mlp/norm":547.7848352225518,"train/train/tensor_act_model_layers_22_self_attn_k_proj/norm":5625.87838774352,"train/train/tensor_act_model_layers_15_self_attn/max_abs":1.046875,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/norm":0.031955911845927325,"train/train/layer__model_layers_72/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_mlp_waleed_W_u_weight/norm":0.07828086563594666,"train/train/tensor_act_model_layers_77/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_mlp_waleed_W_u_weight/std":0.028076171875,"train/train/tensor_act_model_layers_67_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_mlp_waleed_W_g_weight/mean":-0.00022983551025390625,"train/train/tensor_act_model_layers_1_post_attention_layernorm/norm":5792.609130864042,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/std":0.0242919921875,"train/train/tensor_act_model_layers_8_self_attn_v_proj/max_abs":3.21875,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_70/act/std":0.8989523827979937,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/std":0.026123046875,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/norm":8.5625,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/std":0.04296875,"train/train/tensor_act_model_layers_12_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/norm":3.015625,"train/train/tensor_act_model_layers_37/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_84_post_attention_layernorm/norm":5792.611816409759,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_21/grad/mean":-2.5870717259911406e-07,"train/train/tensor_act_model_layers_48_self_attn_v_proj/std":0.5214881063601946,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/mean":-0.000316619873046875,"train/train/tensor_grad_model_layers_63_mlp_waleed_W_u_weight/mean":5.112960934638977e-07,"train/train/tensor_act_model_layers_92_self_attn_v_proj/max_abs":6.25,"train/train/tensor_act_model_layers_72_mlp_down_proj/norm":817.3820887647857,"train/train/tensor_act_model_layers_17_self_attn_k_proj/mean":0.00827789306640625,"train/train/tensor_act_model_layers_64_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_waleed_W_u_weight/mean":1.4126300811767578e-05,"train/train/tensor_act_model_layers_89_mlp_waleed_W_g/std":0.7285182942604889,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/max_abs":0.0003566741943359375,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/mean":1.9348226487636566e-07,"train/train/tensor_grad_model_layers_92_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/norm":0.012131533806520932,"train/train/tensor_act_model_layers_83_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_mlp_waleed_W_g_weight/mean":-2.5203917175531387e-08,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_24/act/std":0.8848829466859557,"train/train/tensor_act_model_layers_84_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_waleed_W_g_weight/max_abs":0.2109375,"train/train/tensor_act_model_layers_15_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_input_layernorm/std":1.000001246749112,"train/train/tensor_act_model_layers_29_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp_waleed/norm":809.2116498526364,"train/train/tensor_act_model_layers_47_mlp_waleed_W_g/mean":-0.00653076171875,"train/train/tensor_act_model_layers_33_mlp_down_proj/norm":268.4124111021993,"train/train/layer_model_layers_14/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/norm":0.026860600761418437,"train/train/tensor_grad_model_layers_43_mlp_waleed_W_g_weight/mean":-3.7369318306446075e-08,"train/train/tensor_act_model_layers_19_self_attn_k_proj/norm":6006.363668076201,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/mean":1.1071097105741501e-07,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_waleed_W_g_weight/max_abs":0.25,"train/train/layer_model_layers_33/act/std":0.8332138318621557,"train/train/layer_model_layers_85/act/norm":25996.93691907533,"train/train/tensor_act_model_layers_84_mlp/std":0.48730588389156154,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/mean":5.116453394293785e-08,"train/train/tensor_act_model_layers_49/norm":15313.527849020249,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/max_abs":0.0009307861328125,"train/train/tensor_act_model_layers_4/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/mean":6.4849853515625e-05,"train/train/tensor_act_model_layers_85_mlp/mean":-0.01092529296875,"train/train/tensor_act_model_layers_62_input_layernorm/std":1.0000012039883612,"train/train/tensor_act_model_norm/std":1.000000045984051,"train/train/tensor_act_model_layers_59_mlp/mean":0.002735137939453125,"train/train/tensor_act_model_layers_8_post_attention_layernorm/mean":-0.004451751708984375,"train/train/tensor_act_model_layers_30_self_attn_k_proj/norm":5503.484922650558,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/std":4.811348962476154e-05,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/std":5.3181691877669846e-05,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/max_abs":0.00032806396484375,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_36_post_attention_layernorm/norm":5792.168579102348,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/max_abs":0.109375,"train/train/tensor_act_model_layers_46_mlp/std":0.07873568217698812,"train/train/layer__model_layers_13/param/std":0.047535091186234286,"train/train/tensor_param_model_layers_78_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_waleed_W_g_weight/std":0.0245361328125,"train/train/tensor_act_model_layers_89_self_attn/max_abs":7.03125,"train/train/layer__model_layers_1/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp_waleed_W_g/std":0.29687505752141763,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/std":0.033447265625,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_75/param/std":0.05541632263854088,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/max_abs":0.203125,"train/train/tensor_act_model_layers_7_input_layernorm/mean":-0.004486083984375,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/norm":0.02856869136309401,"train/train/tensor_act_model_layers_90_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/mean":-0.00013065338134765625,"train/train/tensor_act_model_layers_64_self_attn_v_proj/std":0.45703126362755747,"train/train/layer__model_layers_5/param/mean":0.0015180233674190718,"train/train/tensor_act_model_layers_49_mlp_waleed_W_g/max_abs":3.171875,"train/train/tensor_act_model_layers_51_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp_waleed_W_g/mean":0.0011243820190429688,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp/mean":0.0008726119995117188,"train/train/tensor_grad_model_layers_30_mlp_waleed_W_g_weight/mean":-1.2724194675683975e-07,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/std":5.275134731110685e-05,"train/train/tensor_act_model_layers_83_self_attn_q_proj/max_abs":6,"train/train/tensor_act_model_layers_1_self_attn_v_proj/std":0.3935560751173485,"train/train/tensor_param_model_layers_47_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_56/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/max_abs":0.00128173828125,"train/train/tensor_act_model_layers_77_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_mlp_waleed_W_g_weight/std":0.024658203125,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/norm":4.8125,"train/train/tensor_act_model_layers_87_mlp_down_proj/max_abs":4.8125,"train/train/tensor_param_model_layers_22_mlp_waleed_W_u_weight/norm":4.34375,"train/train/layer_model_layers_79/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_post_attention_layernorm/max_abs":5.125,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/max_abs":0.2255859375,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp_waleed/std":0.09790069645134682,"train/train/tensor_act_model_layers_87_self_attn_o_proj/mean":-0.008819580078125,"train/train/tensor_act_model_layers_87_self_attn_k_proj/std":1.2148509424002818,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/std":0.029052734375,"train/train/tensor_param_model_layers_11_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_mlp_waleed_W_g_weight/norm":6.0625,"train/train/tensor_act_model_layers_34_mlp_waleed_W_u/norm":2226.7648260629403,"train/train/tensor_act_model_layers_37_post_attention_layernorm/mean":0.0021805763244628906,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/std":0.0281982421875,"train/train/tensor_act_model_layers_93_self_attn_v_proj/norm":4404.149019295027,"train/train/tensor_param_model_layers_80_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/norm":0.016115824180853045,"train/train/tensor_act_model_layers_13_mlp_waleed_W_u/mean":0.00022017955780029297,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/mean":-4.661083221435547e-05,"train/train/tensor_act_model_layers_35_mlp_down_proj/mean":0.002880096435546875,"train/train/tensor_act_model_layers_47_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_mlp_waleed_W_u_weight/max_abs":0.0007476806640625,"train/train/tensor_act_model_layers_84_mlp_waleed_W_g/norm":5476.101744733322,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/norm":0.00689009940052341,"train/train/tensor_act_model_layers_62_self_attn_q_proj/std":0.9599625091695095,"train/train/tensor_grad_model_layers_74_mlp_waleed_W_g_weight/mean":7.725611794739962e-08,"train/train/tensor_param_model_layers_15_mlp_waleed_W_g_weight/norm":4.1875,"train/train/tensor_act_model_layers_86_self_attn_v_proj/max_abs":5.3125,"train/train/layer_model_layers_49/act/norm":18985.923042140657,"train/train/tensor_act_model_layers_75_self_attn_o_proj/std":0.2607441840299499,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9/max_abs":26.5,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_83/act/std":1.0676415907144023,"train/train/tensor_act_model_layers_34_post_attention_layernorm/std":1.0000004437896604,"train/train/tensor_act_model_layers_9_self_attn_v_proj/norm":2365.9063639318274,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/std":0.0576171875,"train/train/tensor_act_model_layers_50_mlp_waleed_W_u/max_abs":3.046875,"train/train/tensor_act_model_layers_6_self_attn/std":0.17260802730337652,"train/train/tensor_param_model_layers_20_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/max_abs":0.00113677978515625,"train/train/tensor_act_model_layers_74_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_waleed_W_g_weight/max_abs":0.1865234375,"train/train/layer__model_layers_38/param/norm":20.31726506608665,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/max_abs":0.224609375,"train/train/tensor_grad_model_layers_18_mlp_waleed_W_g_weight/mean":-4.540197551250458e-09,"train/train/tensor_act_model_layers_31_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_waleed_W_g/max_abs":2.65625,"train/train/tensor_act_model_layers_3_mlp_waleed_W_g/std":0.46826252845209115,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/norm":0.015381646514611673,"train/train/tensor_grad_model_layers_9_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/norm":5.40625,"train/train/tensor_act_model_layers_88_self_attn_q_proj/std":1.0586091315362887,"train/train/tensor_param_model_layers_87_mlp_waleed_W_g_weight/std":0.047607421875,"train/train/tensor_param_model_layers_64_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/mean":-1.969747245311737e-07,"train/train/tensor_grad_model_layers_71_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/max_abs":0.1494140625,"train/train/tensor_act_model_layers_54_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/max_abs":0.125,"train/train/tensor_param_model_layers_82_mlp_waleed_W_g_weight/std":0.041015625,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/max_abs":0.0012359619140625,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/std":4.111444784645232e-05,"train/train/tensor_act_model_layers_54_self_attn_v_proj/max_abs":2.4375,"train/train/tensor_act_model_layers_74_input_layernorm/norm":5792.611938481516,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/mean":7.613562047481537e-08,"train/train/tensor_act_model_layers_47_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_3/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/std":0.5830233560260079,"train/train/tensor_act_model_layers_32_mlp_waleed/norm":866.9032409489938,"train/train/tensor_act_model_layers_17_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp_down_proj/norm":2825.899297889857,"train/train/tensor_act_model_layers_43_self_attn/mean":-0.0020961761474609375,"train/train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_o_proj/norm":221.3439311234897,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/max_abs":0.1533203125,"train/train/tensor_grad_model_layers_93_mlp_waleed_W_g_weight/mean":1.9411163520999253e-07,"train/train/tensor_act_model_layers_53_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_mlp_waleed_W_u/mean":-0.0040435791015625,"train/train/tensor_act_model_layers_22_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/norm":0.004265370616384158,"train/train/tensor_act_model_layers_89_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/std":7.05403708580628e-05,"eval/loss":2.2399916648864746,"train/train/layer_model_layers_74/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn/max_abs":1.2890625,"train/train/tensor_act_model_layers_35/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/std":0.0245361328125,"train/train/tensor_param_model_layers_88_mlp_waleed_W_g_weight/norm":8.8125,"train/train/tensor_act_model_layers_62_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp_waleed_W_g/std":0.5625000670552215,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/max_abs":0.000545501708984375,"train/train/tensor_act_model_layers_63_input_layernorm/mean":0.0048370361328125,"train/train/tensor_grad_model_layers_42_mlp_waleed_W_g_weight/max_abs":0.0004425048828125,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/mean":-0.00012159347534179688,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/std":0.00013551407695079745,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_mlp_waleed_W_u_weight/std":0.027099609375,"train/train/tensor_act_model_layers_74_self_attn_v_proj/max_abs":3.859375,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/max_abs":0.0009002685546875,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/norm":4.8125,"train/train/tensor_param_model_layers_32_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_40_mlp_waleed_W_u_weight/norm":0.014315379211307646,"train/train/tensor_param_model_layers_77_mlp_waleed_W_g_weight/mean":0.0002384185791015625,"train/train/tensor_act_model_layers_78_input_layernorm/std":1.000000573461711,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/max_abs":0.0024871826171875,"train/train/tensor_act_model_layers_33_input_layernorm/mean":0.0013778209686279297,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_o_proj/max_abs":0.94140625,"train/train/tensor_act_model_layers_26_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/std":0.0284423828125,"train/train/tensor_param_model_layers_36_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_48_mlp_waleed_W_u_weight/norm":4.90625,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/std":4.9741126791178234e-05,"train/train/tensor_param_model_layers_52_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_93_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/std":6.283454093404922e-05,"train/train/tensor_act_model_layers_43_mlp_down_proj/max_abs":0.76171875,"train/train/tensor_act_model_layers_33_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/global/param/mean":0.0015357858568978851,"train/train/tensor_act_model_layers_7_mlp_waleed/max_abs":3.921875,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/max_abs":0.000637054443359375,"train/train/tensor_param_model_layers_85_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/mean":0.00017261505126953125,"train/train/layer_model_layers_18/grad/max_abs":0.00110626220703125,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_mlp_waleed_W_u_weight/std":0.02294921875,"train/train/tensor_act_model_layers_25/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_q_proj/std":1.1796877210749883,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/norm":0.04377084644117103,"train/train/tensor_act_model_layers_40_self_attn_k_proj/std":0.7900409179778878,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_91_mlp_waleed_W_u_weight/std":0.0001079115673246733,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_72_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp_waleed_W_u/norm":3058.377732694919,"train/train/tensor_act_model_layers_71_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/norm":0.006539116622217723,"train/train/tensor_act_model_layers_9_post_attention_layernorm/norm":5792.613891601947,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/mean":-1.1222437024116516e-07,"train/train/layer__model_layers_54/param/mean":0.0016169198403678336,"train/train/tensor_act_model_layers_0_self_attn/std":0.6455159689969379,"train/train/tensor_param_model_layers_21_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn/mean":0.0009326934814453125,"train/train/tensor_act_model_layers_0_post_attention_layernorm/norm":5792.599121094707,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_16/param/norm":19.275810871805756,"train/train/tensor_act_model_layers_93_mlp_waleed_W_g/max_abs":6.9375,"train/train/tensor_grad_model_layers_63_mlp_waleed_W_g_weight/max_abs":0.00099945068359375,"train/train/tensor_act_model_layers_33_mlp_waleed_W_g/mean":-0.00519561767578125,"train/train/tensor_act_model_layers_57_self_attn_v_proj/mean":0.007171630859375,"train/train/tensor_act_model_layers_52_mlp_waleed/std":0.12451220342667267,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/max_abs":0.00015544891357421875,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/max_abs":0.0001888275146484375,"train/train/tensor_act_model_layers_92_self_attn_o_proj/std":0.6328246332921563,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/max_abs":0.1484375,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/max_abs":0.0002803802490234375,"train/train/tensor_act_model_layers_33_self_attn/mean":9.223818778991699e-05,"train/train/tensor_act_model_layers_68_post_attention_layernorm/norm":5792.611206058097,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/mean":8.58306884765625e-05,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/max_abs":0.2451171875,"train/train/layer_model_layers_60/act/norm":19209.94016309958,"train/train/tensor_param_model_layers_59_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn_q_proj/std":0.8515670632264802,"train/train/tensor_act_model_layers_81_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_q_proj/norm":5964.283032943674,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_waleed_W_g_weight/mean":-0.00014972686767578125,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/mean":-0.000583648681640625,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/mean":2.729211701080203e-08,"train/train/tensor_param_model_layers_16_mlp_waleed_W_u_weight/norm":4.1875,"train/train/tensor_act_model_layers_42/std":2.6250252351171435,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/norm":4.28125,"train/train/layer_model_layers_85/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_61/grad/mean":-9.74083865604795e-08,"train/train/tensor_act_model_layers_65_mlp_waleed_W_g/max_abs":3.234375,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/std":0.00026536403389645783,"train/train/tensor_act_model_layers_65_mlp_waleed/norm":1283.2435217449154,"train/train/tensor_param_model_layers_89_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_input_layernorm/std":1.0000000767176942,"train/train/tensor_grad_model_layers_57_mlp_waleed_W_g_weight/mean":1.2048985809087753e-08,"train/train/tensor_grad_model_layers_12_mlp_waleed_W_u_weight/mean":9.016366675496101e-08,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/std":0.0296630859375,"train/train/tensor_act_model_layers_18_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/norm":0.006378356346913776,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/norm":0.017395340765948345,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_2_self_attn_q_proj/max_abs":4.5625,"train/train/tensor_act_model_layers_7_post_attention_layernorm/mean":-0.0048065185546875,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/std":4.095140078261235e-05,"train/train/tensor_act_model_layers_7_input_layernorm/max_abs":4.84375,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/mean":-0.00013256072998046875,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/norm":3.328125,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/max_abs":0.000675201416015625,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/mean":2.0617153495550156e-07,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/std":0.0556640625,"train/train/tensor_act_model_layers_2_post_attention_layernorm/mean":-0.004791259765625,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/norm":5.125,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_76_mlp_waleed_W_u/max_abs":3.671875,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_v_proj/max_abs":4.3125,"train/train/tensor_act_model_layers_55_post_attention_layernorm/mean":0.0009102821350097656,"train/train/tensor_act_model_layers_55_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_mlp_waleed_W_u_weight/norm":0.016724523812943506,"train/train/tensor_act_model_layers_27_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_mlp_waleed/mean":-0.000797271728515625,"train/train/tensor_act_model_layers_18_mlp_down_proj/mean":0.0003600120544433594,"train/train/tensor_act_model_layers_60_mlp_down_proj/std":0.08667095571559506,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_50/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/mean":0.000629425048828125,"train/train/tensor_param_model_layers_45_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/norm":0.0031930178275885626,"train/train/tensor_act_model_layers_15_self_attn_o_proj/std":0.09155442719583311,"train/train/tensor_act_model_layers_85_self_attn_k_proj/std":1.25000085830659,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/max_abs":0.2197265625,"train/train/tensor_act_model_layers_16_self_attn_o_proj/norm":441.2318672696229,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/norm":3.015625,"train/train/tensor_act_model_layers_5_self_attn_q_proj/max_abs":5.25,"train/train/tensor_act_model_layers_11_mlp_waleed_W_g/norm":2036.3185471401043,"train/train/tensor_act_model_layers_75_self_attn_o_proj/norm":1511.1161012052517,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/mean":-9.72747802734375e-05,"train/train/layer__model_layers_92/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/std":3.003284863431243e-05,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/mean":1.3830140233039856e-07,"train/train/tensor_act_model_layers_58_input_layernorm/norm":5792.607788107272,"train/train/tensor_param_model_layers_40_mlp_waleed_W_g_weight/max_abs":0.1259765625,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp_waleed_W_u/std":0.9746115054755156,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/std":3.1996041919669216e-05,"train/train/tensor_act_model_layers_42_mlp_waleed_W_u/norm":2490.874964849498,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/std":3.328484737332982e-05,"train/train/layer_model_layers_83/grad/frac_near_user_limit":0,"train/train/layer__model_layers_20/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31/max_abs":26.375,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/mean":-0.000209808349609375,"train/train/layer_model_layers_32/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_norm/norm":5792.6090087940365,"train/train/tensor_act_model_layers_46_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_78_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_input_layernorm/max_abs":5.1875,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/norm":0.011905532660113455,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp/std":0.06024204935727695,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn_o_proj/norm":1708.5771977640004,"train/train/tensor_act_model_layers_68_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/mean":1.866137608885765e-07,"train/train/tensor_act_model_layers_23_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/std":2.4439924151174177e-05,"train/train/tensor_grad_model_layers_52_mlp_waleed_W_g_weight/mean":1.1781230568885803e-07,"train/train/tensor_grad_model_layers_48_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/mean":-1.932494342327118e-06,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/std":6.066868234484826e-05,"train/train/tensor_act_model_layers_73_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp_waleed_W_u/max_abs":4.46875,"train/train/tensor_param_model_layers_79_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_mlp_waleed_W_u_weight/mean":0.00031280517578125,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/max_abs":0.302734375,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/mean":3.777677193284035e-08,"train/train/tensor_act_model_layers_91_self_attn_k_proj/mean":0.03350830078125,"train/train/tensor_param_model_layers_72_mlp_waleed_W_u_weight/max_abs":0.1826171875,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/std":0.037109375,"train/train/tensor_act_model_layers_8_post_attention_layernorm/norm":5792.613403326685,"train/train/tensor_act_model_layers_29_self_attn/mean":0.0007238388061523438,"train/train/tensor_grad_model_layers_83_mlp_waleed_W_u_weight/norm":0.02780477963137817,"train/train/tensor_act_model_layers_93_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/mean":2.1711457520723343e-08,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/max_abs":0.1298828125,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_mlp_waleed_W_u_weight/std":4.9294365662109464e-05,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_72/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64/mean":0.0138702392578125,"train/train/tensor_act_model_layers_93/norm":38605.52969497122,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/std":0.00077056884765625,"train/train/layer_model_layers_19/act/max_abs":26.125,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_mlp_waleed_W_u_weight/mean":-1.33514404296875e-05,"train/train/layer_model_layers_78/act/frac_near_user_limit":0,"train/train/layer__model_layers_7/param/mean":0.0015232127839801093,"train/train/tensor_act_model_layers_11_self_attn_q_proj/norm":8142.149914579508,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/max_abs":0.0020751953125,"train/train/tensor_act_model_layers_87_self_attn_v_proj/mean":0.00921630859375,"train/train/tensor_act_model_layers_87_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/mean":-1.6618287190794945e-08,"train/train/tensor_act_model_layers_82_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/norm":0.024722478206847493,"train/train/tensor_grad_model_layers_31_mlp_waleed_W_u_weight/max_abs":0.000774383544921875,"train/train/tensor_act_model_layers_9_mlp_waleed_W_g/norm":2255.452385570616,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/norm":0.0160283247675117,"train/train/tensor_act_model_layers_45_self_attn_k_proj/norm":5784.470586739361,"train/train/tensor_param_model_layers_9_mlp_waleed_W_g_weight/mean":-9.822845458984375e-05,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/norm":0.14869667315478113,"train/train/tensor_act_model_layers_52_mlp/std":0.08142193545229959,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/norm":0.0005551596964650764,"train/train/tensor_param_model_layers_42_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/norm":4.84375,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/mean":0.00010156631469726562,"train/train/tensor_grad_model_layers_20_mlp_waleed_W_u_weight/std":4.3416088030147115e-05,"train/train/tensor_act_model_layers_72_mlp/max_abs":1.2578125,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/norm":5,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/mean":5.190668161958456e-08,"train/train/layer__model_layers_49/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_input_layernorm/max_abs":5.15625,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/mean":1.932494342327118e-08,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/mean":-1.249462366104126e-05,"train/train/layer_model_layers_40/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/norm":0.013764002887355532,"train/train/tensor_act_model_layers_51_self_attn_o_proj/std":0.2949640231345105,"train/train/tensor_param_model_layers_59_mlp_waleed_W_u_weight/mean":-0.000263214111328125,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/mean":-0.00016021728515625,"train/train/tensor_grad_model_layers_47_mlp_waleed_W_u_weight/std":3.67774402484452e-05,"train/train/tensor_param_model_layers_23_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer_model_layers_41/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_waleed_W_u_weight/max_abs":0.0026702880859375,"train/train/tensor_act_model_layers_42_post_attention_layernorm/norm":5792.608642579982,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/max_abs":5.793571472167969e-05,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/max_abs":0.000244140625,"train/train/tensor_act_model_layers_39_self_attn_o_proj/norm":862.0958548397188,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/max_abs":0.00112152099609375,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_23_self_attn/norm":2107.423249407683,"train/train/tensor_act_model_layers_71_mlp_waleed/std":0.18774472112024884,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_input_layernorm_weight/std":0,"train/train/layer__model_layers_39/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_q_proj/mean":-0.05010986328125,"train/train/tensor_act_model_layers_59_mlp_waleed_W_u/max_abs":2.75,"train/train/tensor_param_model_layers_57_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/max_abs":0.00112152099609375,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/mean":4.5318156480789185e-06,"train/train/tensor_act_model_layers_57_self_attn_k_proj/max_abs":4.5625,"train/train/tensor_act_model_layers_65_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/mean":-3.0605588108301163e-07,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_57/act/max_abs":24,"train/train/tensor_act_model_layers_50_self_attn_v_proj/norm":2258.018975050532,"train/train/layer_model_layers_69/act/max_abs":25.375,"train/train/tensor_grad_model_layers_39_mlp_waleed_W_u_weight/std":4.053117153128104e-05,"train/train/tensor_param_model_layers_93_mlp_waleed_W_u_weight/std":0.0693359375,"train/train/layer_model_layers_16/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/mean":-1.339241862297058e-06,"train/train/layer__model_layers_83/param/norm":23.81149522158363,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/mean":-3.266334533691406e-05,"train/train/tensor_act_model_layers_27_mlp_down_proj/max_abs":0.458984375,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/max_abs":0.00098419189453125,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/norm":0.03159859066865569,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/std":4.7340955915856336e-05,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/norm":2.78125,"train/train/tensor_act_model_layers_38_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_7/norm":18437.792670522325,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/std":0.02978515625,"train/train/tensor_act_model_norm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63/max_abs":24.625,"train/train/tensor_act_model_layers_1_post_attention_layernorm/mean":-0.0098724365234375,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/max_abs":0.185546875,"train/train/tensor_act_model_layers_16_mlp_waleed_W_g/norm":2013.158709812627,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/norm":0.004821704912245236,"train/train/tensor_act_model_layers_73_self_attn_q_proj/std":1.0625002454308619,"train/train/tensor_act_lm_head/max_abs":12.375,"train/train/tensor_act_model_layers_36_mlp_waleed_W_u/mean":0.0037384033203125,"train/train/tensor_act_model_layers_33_self_attn_q_proj/mean":0.03509521484375,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/mean":0.00014209747314453125,"train/train/tensor_act_model_layers_50/std":2.625025333234912,"train/train/tensor_act_model_layers_56_mlp_waleed_W_u/std":0.35937507184849254,"train/train/tensor_act_model_layers_57_mlp_waleed/norm":1086.1345344840313,"train/train/tensor_act_model_layers_78_post_attention_layernorm/std":1.0000005472391744,"total_flos":3.6392499412992e+16,"train/train/tensor_act_model_layers_62_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_mlp_waleed_W_g_weight/mean":-8.288770914077759e-08,"train/train/tensor_param_model_layers_88_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_post_attention_layernorm/std":1.0000000769155093,"train/train/tensor_act_model_layers_29_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn_k_proj/norm":5746.5382072807715,"train/train/tensor_act_model_layers_27_self_attn_k_proj/mean":-0.016204833984375,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/mean":0.0004558563232421875,"train/train/tensor_act_model_layers_46_self_attn/norm":512.7288572803968,"train/train/layer_model_layers_52/act/max_abs":23.375,"train/train/tensor_act_model_layers_73_self_attn_q_proj/mean":0.0985107421875,"train/train/tensor_param_model_layers_36_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_23/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/std":4.869770382318854e-05,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/global/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_input_layernorm/mean":0.003612518310546875,"train/train/tensor_act_model_layers_21_self_attn_v_proj/mean":0.00994873046875,"train/train/tensor_grad_model_layers_65_mlp_waleed_W_u_weight/norm":0.017883821551451167,"train/train/tensor_act_model_layers_22_self_attn_k_proj/mean":0.0343017578125,"train/train/tensor_grad_model_norm_weight/norm":0.11047653278738617,"train/train/tensor_grad_model_layers_0_mlp_waleed_W_u_weight/std":0.0014899945808415896,"train/train/tensor_act_model_layers_67_mlp_down_proj/max_abs":1.0859375,"train/train/tensor_act_model_layers_52_post_attention_layernorm/max_abs":5.03125,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53/norm":15127.663377562183,"train/train/tensor_grad_model_layers_69_mlp_waleed_W_u_weight/max_abs":0.000507354736328125,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_19/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp_waleed_W_g/mean":-0.001678466796875,"train/train/layer__model_layers_70/param/max_abs":1,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/mean":-5.238689482212067e-07,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/max_abs":0.0002651214599609375,"train/train/tensor_act_model_layers_70_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/std":0.0400390625,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/max_abs":0.0001201629638671875,"train/train/tensor_act_model_layers_67_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_87/grad/norm":0.07586292585838973,"train/train/tensor_act_model_layers_54_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/max_abs":0.1904296875,"train/train/tensor_param_model_layers_11_mlp_waleed_W_u_weight/mean":0.00022983551025390625,"train/train/tensor_act_model_layers_14_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_mlp_waleed_W_u_weight/norm":9.75,"train/train/layer_model_layers_90/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/std":0.00028869232228294117,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/std":2.3222785405902433e-05,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/norm":3.5,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/std":0.00011327142312241773,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/mean":1.1506490409374237e-06,"train/train/layer_model_layers_23/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_10_self_attn_k_proj/max_abs":8.5,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_mlp_waleed_W_g_weight/mean":0.0001926422119140625,"train/train/tensor_param_model_layers_61_mlp_waleed_W_u_weight/norm":5.40625,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_o_proj/max_abs":0.515625,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/mean":1.0617077350616455e-06,"train/train/tensor_grad_model_layers_21_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn_k_proj/norm":6796.489503221479,"train/train/tensor_act_model_layers_55_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/mean":-1.8496066331863403e-06,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/std":0.0299072265625,"train/train/tensor_act_model_layers_19_self_attn_k_proj/mean":-0.0048370361328125,"train/train/tensor_act_model_layers_21_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/std":3.076184382908257e-05,"train/train/tensor_grad_model_layers_56_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer_model_layers_86/act/std":1.1098013770338182,"train/train/tensor_param_model_layers_28_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_58_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_32/act/std":0.8261553495309809,"train/train/tensor_act_lm_head/mean":-1.970703125,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/norm":0.00156978384651252,"train/train/tensor_act_model_layers_39_post_attention_layernorm/mean":0.0017390251159667969,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_waleed/mean":0.0003314018249511719,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/mean":3.697641659528017e-08,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/mean":1.6934791347011924e-08,"train/train/tensor_grad_model_layers_73_mlp_waleed_W_u_weight/std":6.164005031748148e-05,"train/train/tensor_act_model_layers_52/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/max_abs":0.00109100341796875,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_81_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/norm":0.0016998960457842553,"train/train/tensor_act_model_layers_30_self_attn_k_proj/mean":-0.005584716796875,"train/train/tensor_act_model_layers_71_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_waleed_W_u_weight/mean":-0.00018787384033203125,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/norm":0.019044338079273193,"train/train/tensor_act_model_layers_44_input_layernorm/mean":0.0040950775146484375,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_18/act/norm":20514.273622497974,"train/train/tensor_act_model_layers_84_self_attn/std":0.9571019647708569,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/norm":0.0012410025770637938,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_down_proj/std":1.042976535074239,"train/train/layer_model_layers_32/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_mlp_waleed_W_g_weight/norm":5.875,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_23/act/norm":21662.646025993785,"train/train/tensor_act_model_layers_78_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/max_abs":0.08203125,"train/train/tensor_act_model_layers_69_mlp_waleed/norm":1450.0054756643772,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/max_abs":0.00016021728515625,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/norm":0.009800182234519714,"train/train/layer__model_layers_27/param/mean":0.0015498502020158931,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn/norm":1349.676532873374,"train/train/tensor_act_model_layers_43_mlp_waleed/max_abs":5.53125,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_3_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/std":0.0390625,"train/train/tensor_act_model_layers_29_input_layernorm/max_abs":5.53125,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/norm":0.003817214135527151,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/std":0.02587890625,"train/train/tensor_grad_model_layers_35_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp_waleed_W_u/std":0.3164063923888416,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/std":3.904341842416969e-05,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/norm":3.421875,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/std":4.542293176035392e-05,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_84/grad/std":0.00011551657244223508,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/norm":0.03542689942279754,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/max_abs":0.00048828125,"train/train/tensor_param_model_layers_83_mlp_waleed_W_u_weight/mean":-2.8133392333984375e-05,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/mean":5.122274160385132e-08,"train/train/tensor_act_model_layers_14_self_attn_v_proj/norm":2022.535739784009,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/std":0.03759765625,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_waleed_W_u_weight/mean":-7.581710815429688e-05,"train/train/layer_model_layers_3/act/mean":-0.001510661095380783,"train/train/tensor_grad_model_layers_37_mlp_waleed_W_u_weight/mean":-4.298635758459568e-08,"train/train/tensor_param_model_layers_14_mlp_waleed_W_g_weight/mean":7.343292236328125e-05,"train/train/layer_model_layers_11/grad/mean":-9.003063364538499e-08,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/norm":0.028997031273653544,"train/train/tensor_act_model_layers_79_input_layernorm/max_abs":5.5,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/max_abs":0.00078582763671875,"train/train/tensor_act_model_layers_53/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/max_abs":0.00083160400390625,"train/train/tensor_param_model_layers_15_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_85_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/max_abs":0.0003795623779296875,"train/train/tensor_act_model_layers_13/mean":-0.00757598876953125,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/max_abs":0.318359375,"train/train/tensor_grad_model_embed_tokens_weight/mean":1.0681105777621269e-07,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/norm":4.75,"train/train/tensor_param_model_layers_70_mlp_waleed_W_g_weight/max_abs":0.1669921875,"train/train/tensor_act_model_layers_64_mlp_waleed_W_g/max_abs":3.3125,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/norm":0.0010353934493098679,"train/train/tensor_act_model_layers_64_input_layernorm/mean":0.00502777099609375,"train/train/tensor_act_model_layers_42_mlp_down_proj/norm":360.68942183749544,"train/train/tensor_param_model_layers_12_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_post_attention_layernorm/std":1.0000010751762052,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/mean":0.00015163421630859375,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/mean":8.303322829306126e-08,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_mlp_waleed_W_g/std":0.24609378359228556,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/std":0.041015625,"train/train/tensor_act_model_layers_0/mean":-0.00766754150390625,"train/train/tensor_grad_model_layers_39_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/norm":0.008041627941672487,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/mean":-2.0605511963367462e-07,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/norm":3.171875,"train/train/tensor_act_model_layers_5_self_attn_q_proj/norm":7729.314461349593,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/std":1.4846947359054341e-05,"train/train/tensor_act_model_layers_54_self_attn/max_abs":1.375,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/max_abs":0.00122833251953125,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/mean":-3.041350282728672e-08,"train/train/layer_model_layers_57/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/norm":6.4375,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/std":0.033935546875,"train/train/tensor_grad_model_layers_10_mlp_waleed_W_g_weight/mean":1.0989606380462646e-07,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_2_post_attention_layernorm/norm":5792.610229493426,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/mean":0.000392913818359375,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/max_abs":0.00089263916015625,"train/train/tensor_param_model_layers_92_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_mlp_down_proj/mean":0.0019817352294921875,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp_down_proj/mean":3.059208393096924e-05,"train/train/tensor_param_model_layers_4_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn/norm":144.58706390062105,"train/train/tensor_act_model_layers_67_self_attn_q_proj/std":1.2421875179938549,"train/train/tensor_act_model_layers_9_mlp_waleed/mean":-0.00342559814453125,"train/train/tensor_param_model_layers_30_mlp_waleed_W_u_weight/std":0.024658203125,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/norm":0.0007496972040916433,"train/train/tensor_act_model_layers_41_mlp_waleed_W_u/std":0.3066406576496763,"train/train/tensor_act_model_layers_53_self_attn_o_proj/std":0.15582413819778,"train/train/tensor_param_model_layers_31_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_mlp_waleed_W_u_weight/mean":-0.00017261505126953125,"train/train/tensor_grad_model_layers_50_mlp_waleed_W_u_weight/norm":0.015951938306663947,"train/train/tensor_act_model_layers_80_mlp/norm":1627.3210229970473,"train/train/tensor_act_model_layers_9_mlp_waleed/std":0.13525478955782616,"train/train/tensor_param_model_layers_83_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_65_self_attn_q_proj/max_abs":5.53125,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/mean":-1.4796387404203415e-07,"train/train/tensor_act_model_layers_66/mean":0.017730712890625,"train/train/tensor_act_model_layers_35_mlp_down_proj/max_abs":0.5625,"train/train/layer_model_layers_34/grad/max_abs":0.00115203857421875,"train/train/tensor_grad_model_layers_66_mlp_waleed_W_u_weight/std":5.156092036653589e-05,"train/train/layer_model_layers_8/grad/mean":2.9341492063820644e-07,"train/train/layer_model_layers_15/grad/norm":0.04268402726741883,"train/train/tensor_act_model_layers_15/norm":16959.541104802203,"train/train/tensor_grad_model_layers_75_mlp_waleed_W_g_weight/std":6.509935788223026e-05,"train/train/tensor_act_model_layers_24_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp_down_proj/std":0.08398650451787402,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/max_abs":0.1259765625,"train/train/layer__model_layers_10/param/norm":19.386589275837046,"train/train/tensor_param_model_layers_5_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/std":0.0017342563510628966,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/mean":-8.630752563476562e-05,"train/train/tensor_grad_model_layers_80_mlp_waleed_W_u_weight/mean":-1.2994860298931599e-07,"train/train/tensor_act_model_layers_85_self_attn_v_proj/max_abs":3.5,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/mean":-8.088536560535431e-06,"train/train/tensor_act_model_layers_28_mlp_waleed_W_g/std":0.267578133130378,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/std":2.1809937639429885e-05,"train/train/tensor_act_model_layers_89_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/std":1.8211885151873095e-05,"train/train/tensor_act_model_layers_65_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/norm":0.028038862566209872,"train/train/tensor_act_model_layers_24_mlp/norm":486.1305667937912,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/mean":9.851646609604359e-09,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/norm":0.025132480907298225,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/mean":-2.468004822731018e-07,"train/train/tensor_act_model_layers_3_self_attn_k_proj/mean":0.0096435546875,"train/train/tensor_act_model_layers_20_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_post_attention_layernorm/std":1.0000000216532496,"train/train/tensor_param_model_layers_14_mlp_waleed_W_g_weight/max_abs":0.099609375,"train/train/layer_model_layers_83/grad/max_abs":0.0010833740234375,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/std":0.02587890625,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/norm":4.65625,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/norm":5.75,"train/train/tensor_act_model_layers_63_self_attn_q_proj/std":0.9892592886700025,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/std":0.00014517489158216547,"train/train/tensor_param_model_layers_20_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_13_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/total_time_seconds":2723.8394591100514,"train/train/tensor_act_model_layers_50_post_attention_layernorm/mean":1.811981201171875e-05,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/max_abs":0.1142578125,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60/mean":0.0126953125,"train/train/tensor_act_model_layers_78_self_attn_k_proj/std":0.9472678961185876,"train/train/tensor_act_model_layers_41_input_layernorm/max_abs":5.28125,"train/train/tensor_act_model_layers_75_self_attn_v_proj/max_abs":3.171875,"train/train/tensor_act_model_layers_9_post_attention_layernorm/mean":-0.005035400390625,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/mean":4.069879651069641e-07,"train/train/tensor_act_model_layers_58_mlp_waleed_W_g/std":0.36523441611764035,"train/train/tensor_act_model_layers_23_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_post_attention_layernorm/mean":0.0181884765625,"train/train/tensor_act_model_layers_59_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/norm":0.03296214900894977,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/std":0.0001983902168050358,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/max_abs":0.15625,"train/train/tensor_act_model_layers_52_mlp_waleed_W_u/std":0.34326276897290786,"train/train/tensor_act_model_layers_66_self_attn/mean":0.00237274169921875,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/max_abs":0.00023651123046875,"train/train/tensor_act_model_layers_42_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_waleed_W_g_weight/std":6.039303900424002e-05,"train/train/tensor_act_model_layers_63_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_mlp_waleed_W_g_weight/max_abs":0.0011138916015625,"train/train/tensor_act_model_layers_63_mlp_waleed/max_abs":5.5,"train/train/tensor_act_model_layers_60_mlp_down_proj/norm":502.2470797203119,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/mean":7.2479248046875e-05,"train/train/tensor_act_model_layers_64_mlp_waleed_W_g/std":0.40039071244436564,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/norm":2.921875,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/mean":-1.4551915228366852e-07,"train/train/tensor_act_model_layers_80_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_mlp_waleed_W_g_weight/max_abs":0.26171875,"train/train/layer__model_layers_92/param/mean":0.0016872417908190937,"train/train/tensor_act_model_layers_82_input_layernorm/std":1.000000169035033,"train/train/layer__model_layers_72/param/max_abs":1,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/mean":-8.028000593185425e-07,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_23/grad/max_abs":0.00555419921875,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_mlp_waleed_W_g_weight/norm":0.03614064724497228,"train/train/tensor_grad_model_layers_26_mlp_waleed_W_u_weight/std":4.4885585037366596e-05,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp_waleed_W_u/mean":0.022430419921875,"train/train/tensor_act_model_layers_44_mlp/mean":0.001728057861328125,"train/train/tensor_act_model_layers_25_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/mean":-3.9577484130859375e-05,"train/train/layer_model_layers_37/grad/std":4.22515827860706e-05,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/norm":6.65625,"train/train/tensor_grad_model_layers_29_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer__model_layers_45/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp_down_proj/norm":476.3652992310078,"train_samples_per_second":199.243,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/norm":4.84375,"train/train/layer__model_layers_60/param/max_abs":1,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/norm":4.03125,"train/train/tensor_param_model_layers_32_mlp_waleed_W_g_weight/norm":4.5,"train/train/tensor_grad_model_layers_24_mlp_waleed_W_g_weight/max_abs":0.0009307861328125,"train/train/tensor_param_model_layers_87_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn_o_proj/mean":0.0006399154663085938,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/norm":0.02909098439332173,"train/train/tensor_act_model_layers_70_self_attn_o_proj/std":0.22876453268411048,"train/train/layer_model_layers_90/act/max_abs":35.5,"train/train/layer__model_layers_12/param/norm":19.275354903104247,"train/train/tensor_act_model_layers_1_mlp_waleed_W_g/mean":-0.00548553466796875,"train/train/layer_model_layers_73/act/mean":0.011603504419326782,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/max_abs":0.0004673004150390625,"train/train/tensor_grad_model_layers_77_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/std":0.0299072265625,"train/train/tensor_param_model_layers_41_mlp_waleed_W_u_weight/mean":3.552436828613281e-05,"train/train/tensor_act_model_layers_11/std":3.0039546543560105,"train/train/tensor_grad_model_layers_20_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_mlp_waleed_W_u_weight/std":0.034423828125,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/std":9.106718717270227e-06,"train/train/tensor_act_model_layers_59_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/std":0.0400390625,"train/train/tensor_act_model_layers_21_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_33_self_attn_o_proj/std":0.0572521984113172,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp_waleed_W_g/max_abs":3.796875,"train/train/tensor_act_model_layers_63_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_post_attention_layernorm/norm":5792.604858398729,"train/train/tensor_act_model_layers_30_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/layer_model_layers_77/act/frac_near_user_limit":0,"train/train/layer_model_layers_18/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_mlp_waleed_W_g_weight/std":3.958721098542851e-05,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp_waleed_W_u/norm":2617.7853429231855,"train/train/tensor_act_model_layers_39/max_abs":25.125,"train/train/global/grad/std":0.00034794211709510863,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn_o_proj/norm":1324.7645765981647,"train/train/tensor_param_model_layers_27_mlp_waleed_W_g_weight/max_abs":0.1279296875,"train/train/tensor_act_model_layers_77_self_attn/max_abs":3.59375,"train/train/tensor_act_model_layers_56_post_attention_layernorm/norm":5792.602050787291,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/std":0.034912109375,"train/train/tensor_act_model_layers_80_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_23/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn/norm":1382.6337837232854,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/max_abs":0.1455078125,"train/train/tensor_act_model_layers_92_self_attn/std":0.6328246332921563,"train/train/tensor_act_model_layers_3/mean":-0.02020263671875,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/mean":3.0959199648350477e-09,"train/train/tensor_act_model_layers_71_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp_waleed_W_g/mean":-1.806020736694336e-05,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp/norm":789.2936829990747,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/mean":-1.6007106751203537e-09,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/std":0.024658203125,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/norm":3.25,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_k_proj/max_abs":6.1875,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/mean":2.130866050720215e-06,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/std":5.060727770555895e-05,"train/train/tensor_act_model_layers_45_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/std":0.03955078125,"train/train/tensor_act_model_layers_87_mlp_waleed_W_u/max_abs":4.40625,"train/train/tensor_act_model_layers_42_self_attn/norm":712.9822687184986,"train/train/tensor_act_model_layers_74_mlp/std":0.16796876542096842,"train/train/layer_model_layers_93/act/std":2.211117443792551,"train/train/tensor_act_model_layers_35_mlp_waleed_W_g/mean":-0.014678955078125,"train/train/tensor_act_model_layers_80_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/std":9.004565125462923e-05,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_/std":0,"train/train/tensor_act_model_layers_0_input_layernorm/norm":5792.3681640674085,"train/train/tensor_param_model_layers_55_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/std":7.223492470280426e-05,"train/train/tensor_param_model_layers_17_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_66/param/max_abs":1,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer_model_layers_28/act/mean":-0.002297408878803253,"train/train/tensor_param_model_layers_93_mlp_waleed_W_g_weight/max_abs":0.29296875,"train/train/tensor_act_model_layers_9/std":3.06645370328231,"train/train/tensor_act_model_layers_57_self_attn/max_abs":4.21875,"train/train/tensor_act_model_layers_59_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_q_proj/max_abs":4.75,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/std":0.00020667587195577673,"train/train/tensor_param_model_layers_21_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_91/norm":25405.440511204783,"train/train/tensor_act_model_layers_56_mlp_waleed_W_g/mean":0.006439208984375,"train/train/tensor_act_model_layers_60_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_waleed_W_u_weight/std":0.0252685546875,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_down_proj/norm":491.6653339683475,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/mean":6.198883056640625e-05,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_40_mlp_waleed_W_g_weight/norm":4.78125,"train/train/tensor_act_model_layers_2_self_attn_k_proj/max_abs":4.0625,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/std":6.631132952630027e-05,"train/train/tensor_act_model_layers_72_mlp_waleed_W_u/norm":3621.3102981680913,"train/train/tensor_act_model_layers_15_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/max_abs":0.185546875,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/norm":0.01137176034979119,"train/train/tensor_act_model_layers_74_mlp_down_proj/norm":971.9932275607126,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_mlp_waleed_W_g_weight/max_abs":0.142578125,"train/train/tensor_param_model_layers_42_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn/std":0.5117412452539847,"train/train/layer_model_layers_33/act/max_abs":26.375,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/norm":3.6875,"train/train/tensor_param_model_layers_31_mlp_waleed_W_u_weight/max_abs":0.1298828125,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/mean":1.6557169146835804e-07,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_mlp_waleed_W_u_weight/norm":0.014422356873352735,"train/train/tensor_act_model_layers_18_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer__model_layers_34/param/std":0.04847645289272377,"train/train/tensor_act_model_layers_51_input_layernorm/max_abs":5.125,"train/train/tensor_act_model_layers_69_mlp_waleed_W_u/std":0.42480582450846077,"train/train/tensor_param_model_layers_63_mlp_waleed_W_u_weight/norm":5.59375,"train/train/tensor_act_model_layers_23_mlp_down_proj/std":0.05310128793812479,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/std":4.792309745423541e-05,"train/train/tensor_act_model_layers_23_input_layernorm/mean":-0.00496673583984375,"train/train/tensor_act_model_layers_88_self_attn_q_proj/mean":0.001895904541015625,"train/train/tensor_act_model_layers_13_self_attn_k_proj/std":1.2187500385901862,"train/train/tensor_act_model_layers_31_self_attn_v_proj/max_abs":2.84375,"train/train/layer_model_layers_87/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/norm":0.003520293910476528,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/norm":0.872509272397676,"train/train/tensor_act_model_layers_53_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/std":0.001762163085233812,"train/train/tensor_act_model_layers_36_mlp_waleed/mean":-0.002971649169921875,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/mean":-2.753734588623047e-05,"train/train/tensor_param_model_layers_58_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_46_mlp_waleed/max_abs":3.484375,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/mean":3.337860107421875e-05,"train/train/layer_model_layers_13/grad/std":6.356443567754525e-05,"train/train/layer_model_layers_14/act/std":0.8943057135106578,"train/train/tensor_act_model_layers_64_input_layernorm/std":1.000001159816075,"train/train/layer_model_layers_47/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/mean":-9.881332516670227e-07,"train/train/tensor_act_model_layers_46/std":2.6250256573237047,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/max_abs":0.00011301040649414062,"train/train/tensor_act_model_layers_4/norm":19418.073211935563,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/std":2.8613386158996058e-05,"train/train/tensor_grad_model_layers_56_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn_v_proj/max_abs":2.609375,"train/train/tensor_grad_model_layers_12_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/std":2.4663301139014035e-05,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/norm":0.02771654764950038,"train/train/layer_model_layers_13/grad/norm":0.05153740960528505,"train/train/layer_model_layers_5/grad/max_abs":0.0029449462890625,"train/train/tensor_act_model_layers_32_mlp_waleed_W_g/mean":-0.0051116943359375,"train/train/tensor_grad_model_layers_7_mlp_waleed_W_g_weight/max_abs":0.0022430419921875,"train/train/tensor_act_model_layers_68_mlp_waleed_W_u/max_abs":2.75,"train/train/tensor_act_model_layers_88_post_attention_layernorm/mean":-0.00403594970703125,"train/train/tensor_act_model_layers_70_mlp_waleed_W_g/mean":0.00730133056640625,"train/train/tensor_act_model_layers_34_input_layernorm/max_abs":5.4375,"train/train/tensor_act_model_layers_79_mlp_waleed_W_u/norm":4129.7020777977405,"train/train/tensor_act_model_layers_31_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92/std":4.9531499353617505,"train/train/tensor_act_model_layers_53_post_attention_layernorm/norm":5792.608398440639,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/mean":-0.00019741058349609375,"train/train/tensor_act_model_layers_85_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp_waleed_W_g/std":0.3847659599355992,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/std":0,"train/train/layer__model_layers_27/param/max_abs":1,"train/train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp_down_proj/std":0.04638704720180074,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/std":0.0235595703125,"train/train/tensor_act_model_layers_11_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_waleed_W_u/std":0.4863281326600346,"train/train/layer_model_layers_88/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_post_attention_layernorm/max_abs":5,"train/train/tensor_act_model_layers_14_post_attention_layernorm/max_abs":5.1875,"train/train/tensor_act_model_layers_72/mean":0.00467681884765625,"train/train/tensor_act_model_layers_16_self_attn/norm":441.2318672696229,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_67_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/mean":1.708976924419403e-07,"train/train/tensor_grad_model_layers_31_mlp_waleed_W_u_weight/norm":0.01435932315281587,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/max_abs":0.0030517578125,"train/train/tensor_act_model_layers_50_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_37_self_attn_o_proj/max_abs":1.0390625,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_waleed_W_u_weight/mean":1.0170042514801025e-06,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/norm":0.007905613807669094,"train/train/tensor_act_model_layers_18_mlp_down_proj/norm":387.1145685451204,"train/train/tensor_param_model_layers_49_mlp_waleed_W_g_weight/max_abs":0.1591796875,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/norm":3.015625,"train/train/tensor_act_model_layers_92_self_attn_v_proj/std":0.7363359195308692,"train/train/tensor_act_model_layers_10/std":3.027371621724618,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_input_layernorm/max_abs":5.34375,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/std":4.3514527121986275e-05,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/mean":4.887580871582031e-05,"train/train/tensor_act_model_layers_30_self_attn_q_proj/mean":-0.007843017578125,"train/train/tensor_act_model_layers_20_self_attn/mean":0.00025272369384765625,"train/train/tensor_param_model_layers_27_mlp_waleed_W_u_weight/std":0.0242919921875,"train/train/layer_model_layers_18/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_mlp_waleed_W_u_weight/norm":0.014603832999057815,"train/train/tensor_act_model_layers_53_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_input_layernorm/max_abs":5.4375,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/std":3.021536251423807e-05,"train/train/tensor_act_model_layers_3_self_attn_o_proj/mean":-0.0006875991821289062,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/max_abs":0.1220703125,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_14/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/max_abs":0.11572265625,"train/train/tensor_param_model_layers_8_mlp_waleed_W_g_weight/std":0.02294921875,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_waleed_W_g_weight/max_abs":0.138671875,"train/train/tensor_act_model_layers_19_self_attn_q_proj/mean":-0.0105133056640625,"train/train/layer__model_layers_67/param/max_abs":1,"train/train/tensor_act_model_layers_8_mlp_waleed_W_u/std":0.2812500091838752,"train/train/tensor_act_model_layers_27_post_attention_layernorm/std":1.0000002407777793,"train/train/tensor_act_model_layers_88_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/norm":5792.607910160397,"train/train/tensor_act_model_layers_66_post_attention_layernorm/mean":0.00640106201171875,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/max_abs":0.11962890625,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/max_abs":0.000186920166015625,"train/train/tensor_act_model_layers_30/mean":0.0033111572265625,"train/train/tensor_act_model_layers_22_mlp_waleed_W_g/mean":0.0005192756652832031,"train/train/tensor_act_model_layers_53_input_layernorm/norm":5792.604492188488,"train/train/tensor_act_model_layers_51_self_attn_v_proj/max_abs":4.875,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_mlp_waleed/mean":-0.0032958984375,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/mean":-1.3737007975578308e-07,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/max_abs":0.123046875,"train/train/tensor_act_model_rotary_emb/mean":0.333984375,"train/train/tensor_act_model_layers_58_input_layernorm/std":1.0000009176445164,"train/train/tensor_param_model_layers_21_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/std":0.04052734375,"train/train/tensor_act_model_layers_45_self_attn/std":0.1948249239022955,"train/train/tensor_param_model_layers_56_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_q_proj/std":1.1484378016724484,"train/train/tensor_act_model_layers_4_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_mlp_waleed_W_g/max_abs":2.734375,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/mean":-0.00012302398681640625,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/std":0.03857421875,"train/train/tensor_act_model_layers_32_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer__model_layers_4/param/norm":19.28452286849807,"train/train/tensor_act_model_layers_88_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/mean":9.1552734375e-05,"train/train/tensor_param_model_layers_66_mlp_waleed_W_u_weight/std":0.0322265625,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/max_abs":0.000904083251953125,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/std":6.038313127665113e-05,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/std":0.0003452301025390625,"train/train/tensor_act_model_layers_81_self_attn_k_proj/max_abs":5.75,"train/train/tensor_act_model_layers_15_self_attn_v_proj/std":0.3652343785858409,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/mean":4.861503839492798e-07,"train/train/layer_model_layers_2/grad/mean":-3.2382710675553485e-07,"train/train/tensor_act_model_layers_51_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/norm":4.46875,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/mean":-1.6868580132722855e-07,"train/train/tensor_act_model_layers_65_mlp_waleed/max_abs":4.21875,"train/train/tensor_act_model_layers_9_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_59/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_mlp_waleed_W_g_weight/mean":-6.379559636116028e-08,"train/train/tensor_act_model_layers_91_mlp_waleed_W_g/mean":0.00614166259765625,"train/train/layer__model_layers_29/param/norm":19.63450207861152,"train/train/tensor_act_model_layers_27_self_attn/std":0.23877105521645042,"train/train/tensor_act_model_layers_39_mlp/max_abs":0.63671875,"train/train/tensor_act_model_layers_23_mlp/mean":0.0002722740173339844,"train/train/tensor_act_model_rotary_emb/std":0.71875,"train/train/tensor_act_model_layers_12_self_attn_v_proj/mean":-0.0062255859375,"train/train/tensor_act_model_layers_47_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_51/param/norm":20.544825230456453,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/max_abs":0.2734375,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/std":8.439390551748034e-06,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/max_abs":0.212890625,"train/train/tensor_act_model_layers_22_self_attn_q_proj/mean":0.0042724609375,"train/train/tensor_param_model_layers_68_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_41/param/max_abs":1,"train/train/tensor_act_model_layers_19_mlp_waleed_W_g/std":0.2543959603846086,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/std":0.023193359375,"train/train/tensor_act_model_layers_23_self_attn_q_proj/std":1.3066450166699743,"train/train/tensor_param_model_layers_81_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_67/act/mean":0.009194210171699524,"train/train/tensor_act_model_layers_72_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn/std":0.06812238353290181,"train/train/tensor_act_model_layers_30_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/norm":3.15625,"train/train/tensor_act_model_layers_30_mlp_down_proj/std":0.05993715720573151,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/norm":0.0019508027635918828,"train/train/layer_model_layers_90/grad/std":0.00010385534189541569,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_waleed_W_g/std":0.38085942512903126,"train/train/tensor_act_model_layers_27_self_attn_q_proj/mean":-0.0545654296875,"train/train/layer_model_layers_50/act/max_abs":23.375,"train/train/tensor_act_model_layers_73_self_attn_k_proj/mean":0.0628662109375,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/std":3.67128481601932e-05,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn/std":0.2807666282315268,"train/train/tensor_param_model_layers_43_mlp_waleed_W_u_weight/max_abs":0.1494140625,"train/train/tensor_act_model_layers_77_mlp_waleed_W_g/std":0.49462972600441546,"train/train/tensor_act_model_layers_45_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp_waleed_W_g/max_abs":2.703125,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_post_attention_layernorm/std":1.000000151127085,"train/train/tensor_act_model_layers_66_post_attention_layernorm/max_abs":5.125,"train/train/tensor_act_model_layers_33_self_attn_k_proj/mean":0.05010986328125,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp_waleed/norm":7752.787960395676,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/max_abs":0.1953125,"train/train/layer_model_layers_31/act/max_abs":26.375,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/norm":3.0625,"train/train/tensor_param_model_layers_8_mlp_waleed_W_g_weight/max_abs":0.11083984375,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/std":0.036865234375,"train/train/tensor_act_model_layers_52_self_attn/norm":323.94372028663236,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_post_attention_layernorm/norm":5792.615478521773,"train/train/tensor_act_model_layers_26_mlp_waleed_W_g/std":0.27148442444049387,"train/train/tensor_act_model_layers_27_mlp_waleed/max_abs":2.328125,"train/train/tensor_param_model_layers_54_mlp_waleed_W_u_weight/max_abs":0.14453125,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/mean":0.002658843994140625,"train/train/tensor_act_model_layers_75_self_attn_v_proj/std":0.4589846025121409,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_66/max_abs":24.625,"train/train/tensor_act_model_layers_17_mlp_waleed/mean":-0.00206756591796875,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/max_abs":0.000667572021484375,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/std":0.041015625,"train/train/tensor_act_model_layers_52/norm":15146.037719631922,"train/train/tensor_act_model_layers_38_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/std":6.506221994544942e-05,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_waleed_W_u_weight/norm":5.09375,"train/train/tensor_act_model_layers_24_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_mlp_waleed_W_u_weight/mean":2.0954757928848267e-07,"train/train/tensor_param_model_layers_0_mlp_waleed_W_g_weight/std":0.0277099609375,"train/train/tensor_act_model_layers_87_post_attention_layernorm/std":1.0000001263106162,"train/train/layer_model_layers_27/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_mlp_waleed_W_u_weight/max_abs":0.1845703125,"train/train/tensor_grad_model_layers_70_mlp_waleed_W_u_weight/norm":0.019776332503129154,"train/train/tensor_act_model_layers_74_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/max_abs":0.0003204345703125,"train/train/tensor_act_model_layers_76_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_input_layernorm/norm":5792.6123046890125,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_34/param/mean":0.0014907075164842531,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/std":0.02392578125,"train/train/tensor_act_model_layers_11_input_layernorm/norm":5792.604125977842,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_waleed_W_u/std":0.43750003916543867,"train/train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/max_abs":0.000591278076171875,"train/train/tensor_act_model_layers_24_self_attn_v_proj/mean":-0.002513885498046875,"train/train/tensor_act_model_layers_63_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/mean":-0.00023746490478515625,"train/train/tensor_act_model_layers_61_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_7_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_10/act/std":1.0055284689166384,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/norm":0.003456654622418936,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/std":8.000321083060547e-05,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/mean":-0.00020599365234375,"train/train/tensor_grad_model_layers_54_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_k_proj/mean":0.022857666015625,"train/train/tensor_act_model_layers_63_mlp_waleed/norm":1286.7443269576063,"train/train/tensor_act_model_layers_1_mlp_waleed_W_g/std":0.8720719788587205,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/mean":-1.3253884389996529e-07,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/norm":3.65625,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/max_abs":0.28125,"train/train/tensor_grad_model_layers_21_mlp_waleed_W_u_weight/norm":0.01864951147175978,"train/train/tensor_param_model_layers_60_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_63_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_78_self_attn/norm":1546.8021338793058,"train/train/tensor_act_model_layers_36_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_o_proj/mean":0.00588226318359375,"train/train/tensor_act_model_layers_53_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/max_abs":0.0004367828369140625,"train/train/tensor_grad_model_layers_41_mlp_waleed_W_g_weight/norm":0.015366047584676367,"train/train/tensor_act_model_layers_85_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_mlp_down_proj/mean":-0.02886962890625,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/max_abs":0.00098419189453125,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/std":0.041259765625,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_93/act/mean":-0.005515098571777344,"train/train/tensor_act_model_layers_56_mlp_down_proj/max_abs":0.76953125,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/norm":2.859375,"train/train/tensor_param_model_layers_29_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_82_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp/max_abs":1.671875,"train/train/tensor_act_model_layers_14_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/mean":7.451744750142097e-07,"train/train/tensor_act_model_layers_92_input_layernorm/std":1.0000001152343365,"train/train/tensor_act_model_layers_35_mlp_down_proj/norm":349.48217769225516,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/mean":3.716559149324894e-08,"train/train/tensor_act_model_layers_73_input_layernorm/norm":5792.607299808705,"train/train/tensor_act_model_layers_92_self_attn_o_proj/mean":-0.0114898681640625,"train/train/tensor_act_model_layers_89_mlp/max_abs":5.65625,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/max_abs":0.2490234375,"train/train/tensor_act_model_layers_49_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_mlp_waleed_W_g/max_abs":2.53125,"train/train/tensor_act_model_layers_17/std":2.8516160948498244,"train/train/tensor_act_model_layers_91_self_attn_q_proj/mean":-0.068359375,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/mean":-4.226603778079152e-08,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/std":0.026123046875,"train/train/tensor_param_model_layers_39_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/std":0.023193359375,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/std":7.630367137983095e-05,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/max_abs":0.00095367431640625,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/mean":-4.912726581096649e-07,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_54/param/max_abs":1,"train/train/tensor_act_model_layers_74/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/mean":4.540197551250458e-09,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/norm":0.01783626476894898,"train/train/tensor_act_model_embed_tokens/norm":502.91896806364804,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/max_abs":0.0004215240478515625,"train/train/tensor_act_model_layers_60_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_input_layernorm/mean":-0.00433349609375,"train/train/tensor_act_model_layers_11_mlp_down_proj/std":0.06073007259754281,"train/train/tensor_grad_model_layers_76_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp_waleed/mean":0.006500244140625,"train/train/tensor_grad_model_layers_15_mlp_waleed_W_g_weight/std":4.7821513179735646e-05,"train/train/tensor_grad_model_layers_17_mlp_waleed_W_u_weight/mean":1.1664815247058868e-07,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/norm":0.021706816736231923,"train/train/tensor_act_model_layers_90_self_attn_q_proj/std":1.2695414968223793,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/std":2.640599159852392e-05,"train/train/layer_model_layers_21/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_mlp_waleed_W_g/norm":2147.766345793008,"train/train/tensor_act_model_layers_84_mlp_waleed_W_u/mean":-0.0303955078125,"train/train/tensor_act_model_layers_65/max_abs":24.5,"train/train/tensor_act_model_layers_71_post_attention_layernorm/mean":0.003910064697265625,"train/train/tensor_param_model_layers_19_mlp_waleed_W_u_weight/mean":6.628036499023438e-05,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_k_proj/mean":0.0022045373916625977,"train/train/tensor_act_model_layers_13_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_waleed_W_u/std":0.28759896945136193,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp/mean":-0.00313568115234375,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/std":0.044921875,"train/train/tensor_act_model_layers_16_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/std":0.0400390625,"train/train/tensor_grad_model_layers_28_mlp_waleed_W_u_weight/std":3.9345223243986546e-05,"train/train/tensor_act_model_layers_48_self_attn_o_proj/std":0.5586146459868395,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/std":0.0361328125,"train/train/layer__model_layers_74/param/max_abs":1,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/std":8.072614972487501e-05,"train/train/tensor_act_model_layers_22/norm":15944.599523157829,"train/train/tensor_act_model_layers_51_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_mlp_waleed_W_g_weight/std":6.21084136827309e-05,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/std":0.0001138070342878088,"train/train/tensor_act_model_layers_34_self_attn/std":0.17505070969498088,"train/train/tensor_act_model_layers_72_self_attn/std":0.28662261158282965,"train/train/tensor_act_model_layers_17_input_layernorm/mean":-0.005767822265625,"train/train/layer__model_layers_47/param/norm":19.515393563479574,"train/train/tensor_param_model_layers_37_mlp_waleed_W_g_weight/std":0.0255126953125,"train/train/layer__model_layers_80/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_26_mlp_waleed_W_g_weight/mean":-0.00010061264038085938,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/mean":-3.9371661841869354e-07,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/norm":6.9375,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/std":0.036376953125,"train/train/tensor_act_model_layers_42_post_attention_layernorm/std":1.0000012723822782,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/max_abs":0.244140625,"train/train/tensor_param_model_layers_60_mlp_waleed_W_g_weight/norm":5.28125,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/max_abs":0.00010967254638671875,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_post_attention_layernorm/mean":-0.006195068359375,"train/train/tensor_param_model_layers_44_mlp_waleed_W_g_weight/norm":4.78125,"train/train/tensor_param_model_layers_25_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7/mean":-0.0012431144714355469,"train/train/tensor_act_model_layers_56_self_attn/norm":734.9091898436315,"train/train/tensor_act_model_layers_1_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/std":0.04443359375,"train/train/tensor_act_model_layers_72_mlp_down_proj/mean":-0.000789642333984375,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/max_abs":0.0001277923583984375,"train/train/layer_model_layers_20/act/std":0.88223806384223,"train/train/tensor_act_model_layers_30_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_1/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/norm":5.375,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_mlp_waleed_W_u_weight/norm":12.5,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_42/act/mean":0.002014924306422472,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/mean":-1.959502696990967e-06,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/std":0.044921875,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn_o_proj/std":0.2807666282315268,"train/train/layer__model_layers_75/param/norm":22.455517008980777,"train/train/tensor_act_model_layers_67/std":2.679723664092204,"train/train/tensor_act_model_layers_50_mlp_waleed_W_g/norm":2793.412534056813,"train/train/tensor_act_model_layers_33_mlp_down_proj/mean":0.00066375732421875,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/max_abs":0.00034332275390625,"train/train/layer__model_layers_72/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/mean":-0.0002193450927734375,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/mean":-6.151199340820312e-05,"train/train/tensor_act_model_layers_11/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/std":3.868873180579975e-05,"train/train/tensor_act_model_layers_93_mlp_down_proj/std":2.9258102567072646,"train/train/layer_model_layers_4/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/max_abs":0.11767578125,"train/train/tensor_act_model_layers_23_mlp_waleed_W_u/norm":1917.4748641661529,"train/train/tensor_act_model_layers_82_self_attn_o_proj/mean":0.0014188289642333984,"train/train/layer__model_layers_70/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/norm":4.6875,"train/train/tensor_act_model_layers_80_self_attn_v_proj/mean":0.0106201171875,"train/train/tensor_act_model_layers_44_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/mean":0.000225067138671875,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/mean":-0.00022792816162109375,"train/train/tensor_act_model_layers_43/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_post_attention_layernorm/mean":-0.00543212890625,"train/train/layer__model_layers_12/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/max_abs":0.000274658203125,"train/train/layer_model_layers_79/grad/std":7.995098922879154e-05,"train/train/tensor_act_model_layers_37_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_mlp_waleed_W_u_weight/mean":4.839897155761719e-05,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/mean":5.443580448627472e-07,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/mean":-0.00011157989501953125,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/mean":-6.541609764099121e-06,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_8/grad/std":0.00011066879886201704,"train/train/tensor_grad_model_layers_8_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_18/param/max_abs":1,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_mlp_waleed_W_u_weight/std":4.8105902611706495e-05,"train/train/layer_model_layers_38/grad/max_abs":0.00110626220703125,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_o_proj/max_abs":1.3046875,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/norm":0.0010360397536170105,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/max_abs":0.11474609375,"train/train/tensor_param_model_layers_2_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/std":1.6005752577992726e-05,"train/train/tensor_grad_model_layers_39_mlp_waleed_W_g_weight/mean":-1.6557896742597222e-07,"train/train/layer_model_layers_28/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/mean":-2.2411346435546875e-05,"train/train/tensor_grad_model_layers_22_mlp_waleed_W_g_weight/norm":0.021824570838186987,"train/train/tensor_grad_model_layers_19_mlp_waleed_W_u_weight/norm":0.017735833431553917,"train/train/tensor_act_model_layers_16_mlp_waleed/mean":0.00262451171875,"train/train/layer_model_layers_48/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_self_attn_k_proj/std":0.7578125131345285,"train/train/tensor_act_model_layers_75_self_attn_q_proj/norm":6365.3256691851975,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_q_proj/norm":6814.058548391688,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/max_abs":0.000362396240234375,"train/train/tensor_act_model_layers_45_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/norm":4.40625,"train/train/tensor_act_model_layers_25_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_o_proj/mean":0.0006036758422851562,"train/train/tensor_param_model_layers_81_mlp_waleed_W_g_weight/std":0.040283203125,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/std":3.7731440956563244e-05,"train/train/tensor_act_model_layers_51_self_attn_q_proj/norm":5998.054911965648,"train/train/tensor_grad_model_layers_93_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp_waleed_W_u/std":0.28173956176559367,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/std":3.4869229019262e-05,"train/train/tensor_act_model_layers_46_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/std":9.229310358723408e-05,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_2_mlp_down_proj/norm":3445.2096531453913,"train/train/tensor_act_model_layers_5_mlp/std":0.15747127311071102,"train/train/tensor_act_model_layers_9_mlp_waleed/norm":1109.469858545042,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/norm":0.0011388324919531386,"train/train/tensor_act_model_layers_32_self_attn_k_proj/max_abs":4.5625,"train/train/tensor_act_model_layers_23_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/max_abs":0.21484375,"train/train/tensor_param_model_layers_74_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/max_abs":0.00079345703125,"train/train/tensor_act_model_layers_71_mlp_waleed_W_u/max_abs":3.328125,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/mean":0.0003528594970703125,"train/train/tensor_act_model_layers_40/std":2.632872379643609,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/max_abs":0.00025177001953125,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/std":0.03271484375,"train/train/layer__model_layers_82/param/max_abs":1,"train/train/tensor_act_model_layers_62_mlp/max_abs":0.81640625,"train/train/tensor_act_model_layers_90_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_v_proj/mean":0.004669189453125,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/std":0.044677734375,"train/train/tensor_act_model_layers_16_mlp/max_abs":0.498046875,"train/train/tensor_act_model_layers_25_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_6_self_attn_o_proj/max_abs":1.1171875,"train/train/tensor_grad_model_layers_13_mlp_waleed_W_u_weight/std":7.344764230685991e-05,"train/train/layer__model_layers_74/param/norm":23.030083925704655,"train/train/layer_model_layers_5/grad/norm":0.0864981152281966,"train/train/tensor_act_model_layers_16_post_attention_layernorm/max_abs":5.25,"train/train/tensor_grad_model_layers_92_mlp_waleed_W_u_weight/mean":-2.8387876227498055e-07,"train/train/tensor_act_model_layers_22_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_v_proj/std":0.3945312851689519,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/std":0.046630859375,"train/train/tensor_act_model_layers_43_mlp_waleed_W_u/max_abs":2.984375,"train/train/tensor_act_model_layers_91_self_attn/mean":0.004276275634765625,"train/train/tensor_act_model_layers_26_mlp_waleed/std":0.10571318795953039,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_input_layernorm_weight/std":0,"train/train/layer_model_layers_15/grad/mean":-3.880276141307562e-08,"train/train/tensor_act_model_layers_46_self_attn_v_proj/norm":2005.3968673697982,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/max_abs":0.00012111663818359375,"train/train/tensor_act_model_layers_11_mlp_waleed_W_u/mean":-0.00690460205078125,"train/train/tensor_act_model_layers_6_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/std":0.045166015625,"train/train/tensor_act_model_layers_8_mlp_down_proj/norm":591.0908098819986,"train/train/tensor_param_model_layers_30_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_v_proj/norm":2164.722706709517,"train/train/tensor_act_model_layers_55_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp/max_abs":1.0625,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/norm":0.010412354371425181,"train/train/tensor_grad_model_layers_16_mlp_waleed_W_u_weight/norm":0.01588087381745323,"train/train/tensor_act_model_layers_38_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/std":0.028564453125,"train/train/tensor_param_model_layers_10_mlp_waleed_W_u_weight/mean":0.0002994537353515625,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_18/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_waleed_W_g_weight/mean":0.00010061264038085938,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/mean":0.000396728515625,"train/train/tensor_act_model_layers_28_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp_waleed/mean":0.0012302398681640625,"train/train/tensor_act_model_layers_81_self_attn/max_abs":4.90625,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/mean":-2.6345252990722656e-05,"train/train/tensor_act_model_layers_53_mlp/max_abs":0.74609375,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/std":0.037109375,"train/train/tensor_param_model_layers_31_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_76_mlp_down_proj/mean":-0.00092315673828125,"train/train/tensor_act_model_layers_28_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/mean":-8.344650268554688e-06,"train/train/tensor_act_model_layers_83_self_attn_o_proj/mean":-0.016357421875,"train/train/tensor_act_model_layers_80_self_attn_o_proj/std":0.5400419810718082,"train/train/tensor_act_model_layers_21_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/mean":0.000278472900390625,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/max_abs":0.0005035400390625,"train/train/tensor_act_model_layers_27_mlp_waleed_W_g/mean":0.0011320114135742188,"train/train/tensor_act_model_layers_59_input_layernorm/mean":0.0022687911987304688,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn_o_proj/norm":599.6632659444573,"train/train/tensor_act_model_layers_39_self_attn_v_proj/std":0.44189534127967317,"train/train/tensor_act_model_layers_54_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/std":0.00010952681283170768,"train/train/tensor_act_model_layers_26_self_attn/std":0.07605020564248745,"train/train/tensor_act_model_layers_75_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_v_proj/mean":0.00788116455078125,"train/train/tensor_param_model_layers_56_mlp_waleed_W_g_weight/mean":-1.0609626770019531e-05,"train/train/tensor_act_model_layers_51_input_layernorm/mean":0.0011625289916992188,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_18/param/std":0.04749656828376458,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/global/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_k_proj/std":0.9511740900382035,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/mean":-1.0788789950311184e-07,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/mean":2.773595042526722e-08,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/max_abs":0.0001506805419921875,"train/train/tensor_act_model_layers_12_self_attn/norm":566.6216797264751,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/norm":0.013504330044409086,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/norm":0.0008922376300176193,"train/train/layer_model_layers_26/grad/mean":-5.884513014843237e-08,"train/train/tensor_grad_model_layers_2_mlp_waleed_W_g_weight/norm":0.074427662004731,"train/train/tensor_act_model_layers_80_input_layernorm/mean":0.006378173828125,"train/train/layer__model_layers_5/param/norm":19.280085053948802,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/max_abs":0.189453125,"train/train/layer_model_layers_39/grad/std":4.496388095084279e-05,"train/train/tensor_grad_model_layers_28_mlp_waleed_W_g_weight/norm":0.013921435455165242,"train/train/tensor_act_model_layers_53_mlp/norm":475.5803465210237,"train/train/layer_model_layers_51/act/norm":20000.006469407268,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/max_abs":0.32421875,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/mean":5.967915058135986e-06,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/mean":-3.4226104617118835e-07,"train/train/tensor_act_model_layers_34/max_abs":26,"train/train/tensor_act_model_layers_92_self_attn_v_proj/mean":-0.00502777099609375,"train/train/tensor_act_model_layers_44_self_attn_v_proj/norm":2003.1940018316782,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/max_abs":0.22265625,"train/train/tensor_act_model_layers_52_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_q_proj/max_abs":6.3125,"train/train/tensor_param_model_layers_57_mlp_waleed_W_g_weight/mean":-5.364418029785156e-06,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/max_abs":0.1259765625,"train/train/tensor_param_model_layers_41_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp/std":0.11010773532758321,"train/train/tensor_act_model_layers_16_mlp_waleed_W_u/max_abs":2.203125,"train/train/tensor_act_model_layers_53_self_attn/mean":0.00016305968165397644,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_post_attention_layernorm/mean":0.0094451904296875,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_32_mlp_waleed/std":0.10571317851297744,"train/train/tensor_act_model_layers_86_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/mean":-0.00018024444580078125,"train/train/tensor_act_model_layers_84_post_attention_layernorm/max_abs":5.3125,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_50/grad/max_abs":0.0014190673828125,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/mean":6.628036499023438e-05,"train/train/tensor_grad_model_layers_76_mlp_waleed_W_u_weight/mean":2.3888424038887024e-07,"train/train/tensor_grad_model_layers_6_mlp_waleed_W_u_weight/mean":2.2584572434425354e-07,"train/train/layer_model_layers_59/grad/norm":0.039657287708399276,"train/train/layer_model_layers_37/act/norm":19468.51408467741,"train/train/tensor_grad_model_layers_21_mlp_waleed_W_g_weight/std":4.64563497692695e-05,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/norm":0.02176371947941199,"train/train/tensor_act_model_layers_29_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/norm":0.0016056703811570367,"train/train/tensor_act_model_layers_50_self_attn_q_proj/std":0.9843793005130318,"train/train/layer_model_layers_23/grad/frac_near_dtype_limit":0,"train/train/layer__model_layers_54/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_input_layernorm_weight/std":0.0014190673828125,"train/train/tensor_act_model_layers_60_post_attention_layernorm/max_abs":4.90625,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/std":0.05078125,"train/train/tensor_act_model_layers_73_mlp_waleed/max_abs":4.46875,"train/train/tensor_act_model_layers_40_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_30_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_o_proj/mean":0.002655029296875,"train/train/tensor_act_model_layers_80_input_layernorm/max_abs":5.46875,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_down_proj/mean":0.0009050369262695312,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/layer_model_layers_90/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_mlp_waleed_W_g_weight/std":3.6300974884413466e-05,"train/train/tensor_act_model_layers_35_self_attn_k_proj/max_abs":4.84375,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_mlp_down_proj/mean":0.00220489501953125,"train/train/tensor_act_model_layers_33_self_attn_v_proj/std":0.3344738538226386,"train/train/tensor_param_model_layers_84_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/norm":0.0016607301084861348,"train/train/tensor_act_model_layers_74_self_attn_v_proj/mean":0.0006718635559082031,"train/train/tensor_grad_model_layers_80_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74/mean":0.008941650390625,"train/train/tensor_grad_model_layers_9_mlp_waleed_W_u_weight/std":6.014733303721458e-05,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/max_abs":0.0005950927734375,"train/train/tensor_act_model_layers_93_mlp/std":2.9258102567072646,"train/train/tensor_act_model_layers_49_self_attn/std":0.03150152124348912,"train/train/tensor_act_model_layers_41/std":2.625025280948306,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_mlp_waleed_W_u_weight/mean":-0.0001201629638671875,"train/train/tensor_param_model_layers_43_mlp_waleed_W_u_weight/norm":4.75,"train/train/tensor_act_model_layers_54_mlp/norm":446.69853263602255,"train/train/tensor_act_model_layers_26_post_attention_layernorm/max_abs":5.53125,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/norm":5,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_post_attention_layernorm/mean":0.003814697265625,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/std":0.021728515625,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/max_abs":0.158203125,"train/train/tensor_act_model_layers_81_post_attention_layernorm/norm":5792.609497071788,"train/train/tensor_act_model_layers_70_self_attn_o_proj/max_abs":2.6875,"train/train/tensor_act_/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp_waleed_W_u/mean":-0.004425048828125,"train/train/tensor_act_model_layers_54_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_post_attention_layernorm/norm":5792.608154298116,"train/train/tensor_act_model_layers_88_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/norm":0.004270853012512583,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/mean":5.210749804973602e-07,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp_down_proj/max_abs":0.462890625,"train/train/tensor_param_model_layers_39_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/max_abs":0.00102996826171875,"train/train/tensor_act_model_layers_48_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn_q_proj/mean":0.011077880859375,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_49/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_57/act/mean":0.002730630338191986,"train/train/layer_model_layers_10/grad/std":9.487452154422347e-05,"train/train/tensor_act_model_layers_11_self_attn_o_proj/norm":495.23312504765244,"train/train/layer__model_layers_25/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/std":0.0277099609375,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/std":0.040771484375,"train/train/layer_model_layers_60/act/mean":-0.0007342025637626648,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/std":4.598238556632011e-05,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/std":0.023681640625,"train/train/tensor_grad_model_layers_72_mlp_waleed_W_u_weight/std":5.961124985942355e-05,"train/train/tensor_act_model_layers_90_mlp_waleed/norm":5270.37546498834,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_input_layernorm/mean":0.0016021728515625,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/mean":-3.1257513910532e-08,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/mean":0.0001010894775390625,"train/train/tensor_act_model_layers_21_self_attn_k_proj/std":1.2187500366797808,"train/train/tensor_act_model_layers_48_self_attn/max_abs":4.78125,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/std":0.00011469069790245955,"train/train/tensor_act_model_layers_51_mlp/mean":0.0005474090576171875,"train/train/tensor_param_model_layers_2_mlp_waleed_W_u_weight/max_abs":0.12890625,"train/train/tensor_act_model_layers_91_mlp_waleed/std":0.7841816987644878,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/std":0.049072265625,"train/train/layer_model_layers_39/act/max_abs":25.125,"train/train/tensor_act_model_layers_27_input_layernorm/std":1.000000202468052,"train/train/tensor_act_model_layers_60_self_attn_k_proj/max_abs":4.6875,"train/train/tensor_act_model_layers_36_mlp_waleed_W_g/norm":2365.622854075854,"train/train/tensor_grad_model_layers_73_mlp_waleed_W_g_weight/max_abs":0.00060272216796875,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/norm":4.75,"train/train/tensor_param_model_layers_68_mlp_waleed_W_g_weight/max_abs":0.197265625,"train/train/tensor_act_model_layers_39_self_attn_q_proj/norm":6035.375680980244,"train/train/tensor_param_model_layers_87_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_65/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/max_abs":0.00077056884765625,"train/train/tensor_act_model_layers_85_mlp_waleed/std":0.40723060680514894,"train/train/tensor_act_model_layers_57_self_attn_k_proj/std":0.9697281889311182,"train/train/tensor_act_model_layers_27_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_k_proj/max_abs":5.125,"train/train/tensor_act_model_layers_60_mlp_waleed_W_u/max_abs":2.78125,"train/train/tensor_act_model_layers_93_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_post_attention_layernorm/std":1.0000001137595909,"train/train/layer_model_layers_81/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/mean":-0.004215240478515625,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/norm":0.015623956883897512,"train/train/tensor_act_model_layers_67_mlp_waleed/max_abs":4.78125,"train/train/tensor_act_model_layers_73_self_attn_k_proj/norm":5518.3457313780455,"train/train/tensor_act_model_layers_27_self_attn_q_proj/std":1.144537922445904,"train/train/tensor_act_model_layers_59_mlp_down_proj/mean":0.002735137939453125,"train/train/tensor_act_model_layers_42_self_attn_q_proj/mean":0.00298309326171875,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/mean":1.825392246246338e-07,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/norm":0.015306109486484391,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/max_abs":0.25390625,"train/train/tensor_grad_model_layers_80_mlp_waleed_W_u_weight/std":7.618929579451732e-05,"train/train/tensor_grad_model_layers_67_mlp_waleed_W_u_weight/std":5.5626905067618337e-05,"train/train/tensor_act_model_layers_47_mlp_waleed/max_abs":2.171875,"train/train/layer_model_layers_82/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/std":0.033935546875,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/std":3.0195341835533903e-05,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/norm":0.0013174740591466573,"train/train/tensor_act_model_layers_27_mlp_waleed_W_g/max_abs":1.8515625,"train/train/tensor_act_model_layers_67_self_attn_v_proj/norm":2838.5018203824575,"train/train/tensor_act_model_layers_86_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_k_proj/norm":5023.08327873366,"train/train/tensor_grad_model_layers_6_mlp_waleed_W_u_weight/max_abs":0.002716064453125,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/mean":-7.838010787963867e-06,"train/train/tensor_act_model_layers_12_self_attn_o_proj/mean":-0.001926422119140625,"train/train/tensor_param_model_layers_54_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_waleed_W_u_weight/norm":0.013824833408131487,"train/train/tensor_act_model_layers_3_self_attn_v_proj/std":0.32861438314611524,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/std":6.282748022022529e-05,"train/train/tensor_act_model_layers_30_mlp/norm":347.0232181655934,"train/train/tensor_act_model_layers_85_self_attn/mean":-0.00763702392578125,"train/train/tensor_act_model_layers_46_mlp_waleed/mean":0.001338958740234375,"train/train/tensor_act_model_layers_53_self_attn_v_proj/std":0.418457900876169,"train/train/layer__model_layers_8/param/max_abs":1,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/std":0.0400390625,"train/train/layer__model_layers_26/param/mean":0.0016387368141210024,"train/train/layer_model_layers_51/grad/mean":1.1349131764376219e-09,"train/train/tensor_grad_model_layers_1_mlp_waleed_W_g_weight/max_abs":0.01055908203125,"train/train/tensor_act_model_layers_79/norm":17528.370598141417,"train/train/tensor_param_model_layers_21_mlp_waleed_W_g_weight/mean":5.14984130859375e-05,"train/train/tensor_grad_model_layers_14_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_mlp_waleed_W_u_weight/std":0.0230712890625,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_mlp_waleed/std":0.10217307453633356,"train/train/tensor_act_model_layers_34_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_mlp_waleed_W_g_weight/std":9.370625829172588e-05,"train/train/tensor_act_model_layers_64_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/std":0.033203125,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62/mean":0.01397705078125,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/max_abs":0.0006866455078125,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/mean":4.601478576660156e-05,"train/train/tensor_act_model_layers_60_mlp_down_proj/mean":0.000965118408203125,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/mean":1.516309566795826e-08,"train/train/tensor_act_model_layers_5_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_39/grad/norm":0.03643458906050043,"train/train/tensor_param_model_layers_8_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/mean":-5.2928924560546875e-05,"train/train/tensor_param_model_layers_89_mlp_waleed_W_u_weight/mean":-0.0002384185791015625,"train/train/tensor_act_model_layers_57_mlp_waleed_W_g/mean":0.003650665283203125,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/mean":3.145076334476471e-06,"train/train/tensor_act_model_layers_58_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_20/act/max_abs":26.125,"train/train/tensor_param_model_layers_11_mlp_waleed_W_g_weight/max_abs":0.123046875,"train/train/tensor_act_model_layers_78_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_mlp_waleed_W_u_weight/max_abs":0.00054168701171875,"train/train/layer_model_layers_35/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_26/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_mlp_waleed_W_g/std":0.25830222648958134,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_mlp_waleed_W_g_weight/mean":2.7584974304772913e-08,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_48/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_mlp_waleed_W_u_weight/norm":4.375,"train/train/tensor_act_model_layers_3_self_attn_o_proj/std":0.03816298857885024,"train/train/tensor_act_model_layers_70_self_attn_v_proj/std":0.4638683810971614,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/max_abs":0.177734375,"train/train/layer_model_layers_38/act/mean":-0.003530215471982956,"train/train/tensor_act_model_layers_81_self_attn_k_proj/norm":6913.54608891881,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/std":0.043212890625,"train/train/tensor_param_model_layers_44_input_layernorm_weight/std":0,"train/train/layer__model_layers_53/param/std":0.04977000624986619,"train/train/tensor_act_model_layers_88_self_attn_v_proj/max_abs":4.15625,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/std":0.02685546875,"train/train/tensor_grad_model_layers_47_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/std":0.024658203125,"train/train/tensor_param_model_layers_48_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_2_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/norm":0.0032523689204673094,"train/train/tensor_act_model_layers_63_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_q_proj/std":1.1953125563906677,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_input_layernorm/std":1.0000013364665556,"train/train/tensor_act_model_layers_85_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train_runtime":3854.5847,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_input_layernorm/std":1.000001305735225,"train/train/tensor_act_model_layers_48_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_36/act/mean":0.002873707562685013,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/max_abs":9.202957153320312e-05,"train/train/tensor_act_model_layers_26_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_post_attention_layernorm/max_abs":5.5625,"train/train/tensor_act_model_layers_44_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/std":4.780614779892442e-05,"train/train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_65/grad/norm":0.04259528871039976,"train/train/tensor_param_model_layers_31_mlp_waleed_W_g_weight/std":0.024658203125,"train/train/tensor_act_model_layers_91_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59/norm":15038.59466340581,"train/train/tensor_act_model_layers_52_self_attn_o_proj/max_abs":0.8359375,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_16/act/mean":-0.002299934974871576,"train/train/tensor_act_model_layers_81/norm":18535.469824789703,"train/train/tensor_act_model_layers_43_mlp_waleed/norm":870.5730893924629,"train/train/tensor_act_model_layers_10_mlp/max_abs":0.9296875,"train/train/layer__model_layers_78/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn_q_proj/mean":-0.04522705078125,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/mean":1.0814983397722244e-07,"train/train/tensor_act_model_layers_14_input_layernorm/norm":5792.612304691173,"train/train/tensor_act_model_layers_13_mlp_down_proj/norm":535.5182471211212,"train/train/tensor_param_model_layers_17_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_27_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_post_attention_layernorm/std":1.000001491199685,"train/train/tensor_act_model_layers_12_mlp_down_proj/norm":409.22641123939337,"train/train/tensor_grad_model_layers_11_mlp_waleed_W_g_weight/max_abs":0.00080108642578125,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/norm":4.40625,"train/train/tensor_act_model_layers_51_mlp_waleed/max_abs":3.5,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/mean":1.737847924232483e-06,"train/train/tensor_act_model_layers_66_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn/std":0.24487368773690996,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/mean":-1.0590883903205395e-07,"train/train/tensor_act_model_layers_86_mlp/mean":-0.00231170654296875,"train/train/tensor_act_model_layers_62_post_attention_layernorm/std":1.0000011974109035,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_60/grad/mean":8.528284995864595e-09,"train/train/tensor_act_model_layers_90_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/mean":2.3515895009040833e-08,"train/train/tensor_act_model_layers_26_mlp_waleed_W_u/norm":2198.877853057791,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/norm":0.0022903564371474137,"train/train/tensor_param_model_layers_75_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/norm":0.015856253452999836,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer_model_layers_27/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_v_proj/mean":0.017578125,"train/train/tensor_act_model_layers_69_self_attn_q_proj/max_abs":5.875,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_waleed_W_u/max_abs":2.71875,"train/train/tensor_grad_model_layers_31_mlp_waleed_W_g_weight/std":3.850627054833231e-05,"train/train/tensor_act_model_layers_37_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_post_attention_layernorm/std":1.000001159242639,"train/train/tensor_act_model_layers_73_mlp_down_proj/max_abs":1.484375,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_mlp_waleed_W_g_weight/std":0.023193359375,"train/train/tensor_act_model_layers_54_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/max_abs":0.09521484375,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/mean":4.109460860490799e-08,"train/train/tensor_act_model_layers_42_mlp_down_proj/mean":-4.5027583837509155e-05,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_k_proj/norm":6482.322070910952,"train/train/tensor_grad_model_layers_6_mlp_waleed_W_g_weight/std":0.00011326313324520991,"train/train/layer_model_layers_76/act/max_abs":26,"train/train/tensor_grad_model_layers_11_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/mean":9.202957153320312e-05,"train/train/tensor_param_model_layers_39_mlp_waleed_W_g_weight/norm":4.71875,"train/train/tensor_act_model_layers_90_self_attn_o_proj/mean":0.0049285888671875,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_11/max_abs":26.25,"train/train/tensor_act_model_layers_76_self_attn_o_proj/max_abs":2.359375,"train/train/tensor_act_model_layers_83_mlp/mean":0.00528717041015625,"train/train/tensor_param_model_layers_52_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_45/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/std":8.2665586917365e-05,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/max_abs":0.322265625,"train/train/layer__model_layers_77/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn/max_abs":0.80078125,"train/train/tensor_act_model_layers_88_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_80/grad/max_abs":0.00133514404296875,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/norm":4.84375,"train/train/layer_model_layers_72/grad/max_abs":0.0024871826171875,"train/train/tensor_act_model_layers_18_mlp/std":0.06689453480364134,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/norm":0.013490219383513087,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn_v_proj/norm":2142.819433179917,"train/train/tensor_act_model_layers_52_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/std":0.034423828125,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_75_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/max_abs":0.23828125,"train/train/tensor_grad_model_layers_53_mlp_waleed_W_g_weight/norm":0.015397713524662883,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_1_self_attn_q_proj/norm":5008.96416806064,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_mlp_waleed_W_u_weight/std":0.0230712890625,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/norm":0.0008093813191262935,"train/train/tensor_act_model_layers_60_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/std":0.0260009765625,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/max_abs":0.000797271728515625,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/mean":1.0104849934577942e-07,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/mean":3.83937731385231e-07,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/max_abs":0.228515625,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_13/act/mean":-0.0014766603708267212,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_post_attention_layernorm/mean":-0.00042629241943359375,"train/train/tensor_param_model_layers_80_mlp_waleed_W_g_weight/std":0.0390625,"train/train/tensor_act_model_layers_88_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_waleed_W_u_weight/max_abs":0.0986328125,"train/train/layer_model_layers_33/act/mean":0.005874495953321457,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/mean":0.00019168853759765625,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/mean":0.0002040863037109375,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/max_abs":0.26953125,"train/train/global/act/norm":237550.51854922003,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_input_layernorm/std":1.000000851961523,"train/train/tensor_act_model_layers_50_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/mean":-6.920163286849856e-08,"train/train/tensor_act_model_layers_35_mlp_down_proj/std":0.06024204935727695,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp_waleed/max_abs":5.125,"train/train/tensor_act_model_layers_22_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/max_abs":0.000579833984375,"train/train/tensor_act_model_layers_45/mean":0.0110626220703125,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/std":0.00013186722341145267,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/max_abs":0.000728607177734375,"train/train/tensor_act_model_layers_93_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_o_proj/mean":-0.00021195411682128906,"train/train/tensor_act_model_layers_11_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_mlp_waleed_W_g_weight/norm":0.017311511448754716,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/mean":-5.383044481277466e-06,"train/train/tensor_grad_model_layers_73_mlp_waleed_W_g_weight/mean":1.5464320313185453e-07,"train/train/tensor_act_model_layers_67_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_waleed_W_u_weight/max_abs":0.14453125,"train/train/tensor_grad_model_layers_3_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/mean":3.1245872378349304e-07,"train/train/tensor_act_model_layers_7_mlp_waleed/mean":0.003490447998046875,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/std":0.00013101244220799267,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_mlp_waleed_W_u_weight/mean":8.396455086767673e-09,"train/train/tensor_grad_model_layers_84_mlp_waleed_W_u_weight/max_abs":0.00095367431640625,"train/train/layer_model_layers_88/act/std":1.1661988760823105,"train/train/tensor_act_model_layers_46_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_87/grad/std":9.362725478681353e-05,"train/train/tensor_act_model_layers_22_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn/mean":-0.0005655288696289062,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/mean":-6.628036499023438e-05,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/std":0.0238037109375,"train/train/tensor_act_model_layers_86_input_layernorm/mean":0.003643035888671875,"train/train/tensor_act_model_layers_32_mlp_waleed_W_u/max_abs":2.015625,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/std":0.04736328125,"train/train/tensor_grad_model_layers_16_mlp_waleed_W_g_weight/std":4.3817323826691655e-05,"train/train/tensor_act_model_layers_88_input_layernorm/norm":5792.610595710148,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/std":4.1787087814271566e-05,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/norm":3.375,"train/train/layer_model_layers_9/grad/mean":5.411877705135881e-08,"train/train/tensor_act_model_layers_10_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_input_layernorm/std":1.0000000201107468,"train/train/tensor_act_model_layers_40/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn_v_proj/mean":0.0079193115234375,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/mean":6.058812141418457e-05,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_waleed_W_u_weight/max_abs":0.11279296875,"train/train/tensor_act_model_layers_81_mlp_waleed_W_g/std":0.5742192722499834,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_waleed_W_g_weight/max_abs":0.1796875,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/max_abs":0.0869140625,"train/train/tensor_act_model_layers_64_mlp_down_proj/max_abs":0.98828125,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp_waleed_W_g/norm":4055.024451351985,"train/train/tensor_act_model_layers_62_self_attn_o_proj/max_abs":1.265625,"train/train/layer_model_layers_67/act/std":0.9159738889194162,"train/train/tensor_param_model_layers_65_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/max_abs":0.3203125,"train/train/tensor_act_model_layers_91_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_waleed_W_u_weight/mean":-8.392333984375e-05,"train/train/tensor_grad_model_layers_25_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_mlp_waleed_W_u_weight/std":4.137472805921758e-05,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_q_proj/max_abs":6.25,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/max_abs":0.259765625,"train/train/layer__model_layers_78/param/std":0.05673179665841654,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/std":4.083596034125039e-05,"train/train/tensor_act_model_layers_80_self_attn_o_proj/mean":0.0094451904296875,"train/train/layer_model_layers_6/grad/max_abs":0.00494384765625,"train/train/layer_model_layers_35/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_mlp_waleed_W_g/mean":-0.002315521240234375,"train/train/tensor_act_model_layers_69_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_v_proj/mean":-2.086162567138672e-06,"train/train/tensor_act_model_layers_51_self_attn_v_proj/std":0.47314536280787767,"train/train/tensor_act_model_layers_88_mlp_waleed_W_u/std":0.722656331674468,"train/train/tensor_act_model_layers_45_mlp_waleed_W_g/mean":0.0018482208251953125,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/max_abs":0.00010204315185546875,"train/train/tensor_act_model_layers_73_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/max_abs":0.0015106201171875,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp_waleed_W_g/std":0.26953140048785124,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/std":4.446271307184606e-05,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/mean":-2.2730091586709023e-08,"train/train/tensor_param_model_layers_79_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/norm":5.4375,"train/train/tensor_act_model_layers_41_input_layernorm/std":1.0000012154020514,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/mean":-0.000301361083984375,"train/train/tensor_act_model_layers_38_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_56/act/mean":-0.004194110631942749,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/mean":-3.4335535019636154e-06,"train/train/tensor_param_model_embed_tokens_weight/max_abs":0.5,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/mean":-6.277114152908325e-07,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/norm":0.008386575283641064,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/std":0.0001701319213638586,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/max_abs":0.232421875,"train/train/layer__model_layers_73/param/std":0.05558694676200754,"train/train/tensor_act_model_layers_68_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_waleed_W_u/std":0.4101562871819434,"train/train/tensor_act_model_layers_75_mlp_waleed_W_g/std":0.4843753602714891,"train/train/layer_model_layers_6/act/norm":25549.016608813756,"train/train/tensor_act_model_layers_78_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/max_abs":0.0001964569091796875,"train/train/tensor_act_model_layers_68/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_mlp_waleed_W_u_weight/norm":4.875,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/std":0.00016377841254361765,"train/train/layer_model_layers_33/grad/frac_near_dtype_limit":0,"train/train/tensor_act_lm_head/std":1.843761379400795,"train/train/tensor_param_model_layers_88_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_39_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_26/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81/mean":0.0577392578125,"train/train/tensor_act_model_layers_90_mlp_down_proj/max_abs":7.125,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25/std":2.7578682211685104,"train/train/tensor_act_model_layers_27_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/mean":3.5762786865234375e-07,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_o_proj/max_abs":0.98828125,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_38/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp_waleed/max_abs":9.4375,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_waleed_W_u/std":0.34570369528438066,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/std":0.03857421875,"train/train/tensor_act_model_layers_33_mlp_waleed/frac_near_dtype_limit":0,"train/train/layer_model_layers_72/act/std":0.917389627702202,"train/train/layer__model_layers_48/param/norm":20.005968347751878,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/mean":-2.280808985233307e-06,"train/train/layer_model_layers_83/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/mean":-4.291723598726094e-08,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_o_proj/max_abs":3.59375,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/max_abs":9.679794311523438e-05,"train/train/tensor_act_model_layers_53_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/mean":-7.07223080098629e-08,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/max_abs":0.181640625,"train/train/tensor_act_model_layers_26_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_post_attention_layernorm/mean":0.00551605224609375,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/max_abs":0.26171875,"train/train/layer_model_layers_91/grad/max_abs":0.00174713134765625,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_37/grad/mean":-7.320734833522632e-08,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/max_abs":0.00063323974609375,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/max_abs":0.00054931640625,"train/train/tensor_act_model_layers_77_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/max_abs":0.2421875,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/std":0.00024728388888068466,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/std":0.022705078125,"train/train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/std":0.03857421875,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51/std":2.6250253906707663,"train/train/tensor_param_model_layers_30_mlp_waleed_W_u_weight/mean":3.147125244140625e-05,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp_waleed_W_g/mean":0.002490997314453125,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/mean":-1.31258275359869e-08,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_mlp_waleed_W_g_weight/mean":-5.005858838558197e-07,"train/train/tensor_act_model_layers_83/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_mlp_waleed_W_g_weight/mean":-0.00014972686767578125,"train/train/layer_model_layers_56/act/norm":19281.420288994355,"train/train/layer__model_layers_91/param/std":0.06730397734100878,"train/train/tensor_grad_model_layers_43_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/max_abs":0.1455078125,"train/train/tensor_act_model_layers_70_self_attn_q_proj/norm":6602.840996736434,"train/train/tensor_grad_model_layers_71_mlp_waleed_W_g_weight/std":5.597317740809307e-05,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/max_abs":0.251953125,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_26_mlp_waleed_W_g_weight/max_abs":0.1220703125,"train/train/tensor_act_model_layers_92_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp_waleed/mean":0.0006580352783203125,"train/train/layer_model_layers_86/act/mean":0.011330574750900269,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/mean":-1.5832483768463135e-08,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/mean":2.558808773756027e-07,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp_waleed_W_u/norm":2891.422438188648,"train/train/tensor_act_model_layers_15_post_attention_layernorm/max_abs":5.125,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/norm":5.8125,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/norm":0.015779830802707215,"train/train/tensor_param_model_layers_85_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_v_proj/max_abs":4.9375,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/norm":0.005569876041139503,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/max_abs":0.0006256103515625,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/norm":5.1875,"train/train/tensor_act_model_layers_14_mlp_down_proj/max_abs":0.51953125,"train/train/tensor_act_model_layers_28_mlp_waleed_W_g/mean":-0.002559661865234375,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/max_abs":0.017333984375,"train/train/tensor_act_model_layers_9_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71/norm":15847.77855204111,"train/train/tensor_act_model_layers_76_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn/std":0.08642889780200759,"train/train/tensor_param_model_layers_12_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/std":6.702494327499385e-05,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/norm":0.003297425728602992,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/std":1.865879806115068e-05,"train/train/tensor_act_model_layers_85_self_attn_v_proj/mean":0.012451171875,"train/train/tensor_param_model_layers_81_mlp_waleed_W_g_weight/max_abs":0.2177734375,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_92/param/norm":27.490658924896653,"train/train/tensor_act_model_layers_10_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_4_self_attn/mean":0.0006647109985351562,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_68/grad/mean":-3.434415447544568e-08,"train/train/tensor_act_model_layers_17_post_attention_layernorm/norm":5792.374023440041,"train/train/tensor_act_model_layers_50_mlp_waleed/norm":1080.7532844272514,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/std":0.0001437781473569714,"train/train/layer_model_layers_40/act/max_abs":25,"train/train/layer_model_layers_74/grad/frac_near_user_limit":0,"train/train/layer_model_layers_76/act/mean":0.0025805234909057617,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/norm":0.021551201666056525,"train/train/tensor_param_model_layers_48_mlp_waleed_W_u_weight/max_abs":0.158203125,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/mean":-1.6978010535240173e-06,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_post_attention_layernorm/norm":5792.610961922833,"train/train/tensor_act_model_layers_4_mlp/max_abs":1.3984375,"train/train/tensor_act_model_layers_83_mlp_waleed_W_u/max_abs":3.984375,"train/train/tensor_act_model_layers_82_self_attn/max_abs":4.0625,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/norm":0.014421259057891163,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_mlp_waleed_W_u/std":0.5820317452383814,"train/train/tensor_act_model_layers_40_self_attn_o_proj/max_abs":1.34375,"train/train/tensor_act_model_layers_49_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_mlp_waleed/norm":843.6223939056999,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/max_abs":0.0002956390380859375,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/std":4.022353869689989e-05,"train/train/layer_model_layers_43/grad/mean":-2.267465355708707e-08,"train/train/layer__model_layers_16/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_69/act/std":0.8820190581022099,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/norm":0.0010114704148566737,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/std":2.72972717064997e-05,"train/train/tensor_grad_model_layers_46_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/max_abs":0.000499725341796875,"train/train/tensor_act_model_layers_81_mlp/std":0.3066407095664509,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/max_abs":0.00116729736328125,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/mean":-1.1185184121131897e-06,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_56_mlp_waleed/norm":1049.3706694551188,"train/train/tensor_param_model_layers_31_mlp_waleed_W_g_weight/mean":0.00014209747314453125,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp/norm":3382.6723212187767,"train/train/tensor_act_model_layers_49_mlp_waleed_W_u/norm":2826.0484137382928,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/norm":0.023104124570234343,"train/train/tensor_act_model_layers_11/mean":-0.0070953369140625,"train/train/tensor_act_model_layers_14_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_waleed_W_u/max_abs":2.359375,"train/train/layer_model_layers_47/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/std":8.994989722129009e-05,"train/train/tensor_act_model_layers_92_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/std":0.0002249611244592105,"train/train/tensor_grad_model_layers_50_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_input_layernorm/norm":5792.612182618437,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/std":0.04541015625,"train/train/tensor_act_model_layers_91_mlp/norm":6043.683450899033,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/mean":-0.0002803802490234375,"train/train/tensor_param_model_layers_34_mlp_waleed_W_g_weight/mean":-5.269050598144531e-05,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/std":1.793525626666354e-05,"train/train/tensor_act_model_layers_12/norm":17271.504191882057,"train/train/layer_model_layers_92/grad/max_abs":0.0025787353515625,"train/train/tensor_grad_model_layers_79_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn/norm":2313.196253718315,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp_down_proj/max_abs":0.7109375,"train/train/layer_model_layers_48/grad/norm":0.05405671545522121,"train/train/tensor_act_model_layers_64_mlp_waleed_W_g/norm":3279.9430414813355,"train/train/tensor_act_model_layers_49_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/norm":3.515625,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/norm":0.0006153941467380399,"train/train/tensor_act_model_layers_35_mlp_waleed_W_u/max_abs":2.203125,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/std":0.031982421875,"train/train/tensor_grad_model_layers_42_mlp_waleed_W_u_weight/mean":5.291076377034187e-08,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_41/act/mean":0.00534704327583313,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/norm":5.90625,"train/train/layer__model_layers_80/param/std":0.058099499646465005,"train/train/tensor_grad_model_layers_66_mlp_waleed_W_u_weight/norm":0.018647114285275792,"train/train/tensor_act_model_layers_6_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/mean":0.000141143798828125,"train/train/tensor_act_model_layers_42_post_attention_layernorm/max_abs":5.1875,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/mean":3.334134817123413e-07,"train/train/tensor_act_model_layers_31_mlp_waleed_W_u/std":0.2812500182319323,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/norm":3.65625,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp_waleed_W_g/mean":-0.0097198486328125,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/std":0.00014404953963878734,"train/train/tensor_param_model_layers_38_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_45/param/max_abs":1,"train/train/tensor_act_model_layers_62_self_attn_q_proj/norm":5555.876643544285,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/max_abs":0.000194549560546875,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/mean":0.0001850128173828125,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/norm":0.001004897628286631,"train/train/tensor_act_model_layers_60_self_attn_v_proj/norm":2000.448465613929,"train/train/tensor_act_model_layers_78_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/mean":-0.0001621246337890625,"train/train/tensor_act_model_layers_9_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp_down_proj/std":1.3632870485461193,"train/train/tensor_act_model_layers_54_post_attention_layernorm/norm":5792.61523438423,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/mean":1.3602402759715915e-08,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/mean":0.0002803802490234375,"train/train/tensor_param_model_layers_45_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_k_proj/max_abs":4.59375,"train/train/layer_model_layers_16/grad/norm":0.04182546044193737,"train/train/tensor_act_model_layers_63_mlp/norm":587.1392161318835,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/max_abs":0.09912109375,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_k_proj/std":1.0761773218338997,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_post_attention_layernorm/mean":-0.0072174072265625,"train/train/layer_model_layers_43/grad/std":4.127300611866032e-05,"train/train/layer__model_layers_87/param/std":0.06292232742134558,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21/max_abs":26.375,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/mean":3.5727862268686295e-07,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/layer_model_layers_46/grad/norm":0.03373890574319406,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/max_abs":0.000659942626953125,"train/train/tensor_act_model_layers_34_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/std":0.039794921875,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp_waleed_W_u/mean":-0.0014896392822265625,"train/train/tensor_act_model_layers_46_post_attention_layernorm/mean":0.006175994873046875,"train/train/tensor_param_model_layers_37_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/norm":8.9375,"train/train/layer_model_layers_44/act/frac_near_user_limit":0,"train/train/layer__model_layers_74/param/mean":0.001520165787100234,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/norm":0.0954717837287137,"train/train/tensor_param_model_layers_59_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/std":0.04150390625,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp/max_abs":10.9375,"train/train/tensor_act_model_layers_76_mlp_waleed_W_g/norm":4111.935259822197,"train/train/tensor_grad_model_layers_33_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/norm":0.016202764863929322,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/max_abs":0.09912109375,"train/train/tensor_act_model_layers_25_self_attn_o_proj/max_abs":0.828125,"train/train/layer_model_layers_65/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_mlp_waleed_W_u_weight/std":0.0390625,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/std":0.0240478515625,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp_waleed_W_u/norm":4146.38424741334,"train/train/tensor_act_model_layers_21_self_attn_q_proj/max_abs":6.40625,"train/train/tensor_grad_model_layers_56_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_input_layernorm/std":1.0000002587553232,"train/train/tensor_act_model_layers_85_self_attn_k_proj/max_abs":5.59375,"train/train/layer_model_layers_29/grad/max_abs":0.00119781494140625,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/mean":6.961822509765625e-05,"train/train/tensor_act_model_layers_12_input_layernorm/std":1.0000000317522786,"train/train/tensor_act_model_layers_8_self_attn_q_proj/max_abs":8.5625,"train/train/tensor_act_model_layers_64_mlp/norm":601.7636948557183,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_74/param/std":0.05681743571480268,"train/train/tensor_act_model_layers_2_mlp_waleed/norm":3582.4336628560113,"train/train/tensor_act_model_layers_15_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/norm":5.125,"train/train/tensor_param_model_layers_67_mlp_waleed_W_g_weight/mean":2.4437904357910156e-05,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/std":3.249151129989614e-05,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/norm":0.0008036708016414823,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/std":8.267229686263865e-05,"train/train/tensor_act_model_layers_56_mlp_waleed_W_g/max_abs":2.65625,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/max_abs":0.00136566162109375,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/max_abs":0.21484375,"train/train/layer_model_layers_44/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn_o_proj/std":0.4335939456615995,"train/train/tensor_param_model_layers_31_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer_model_layers_69/grad/norm":0.048449581838807904,"train/train/tensor_act_model_layers_8_self_attn_k_proj/norm":12097.872824997587,"train/train/tensor_act_model_layers_15_self_attn_k_proj/std":1.0468757709402032,"train/train/tensor_act_model_layers_58_self_attn_q_proj/norm":6521.236944352231,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/std":0.0262451171875,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/mean":-0.000324249267578125,"train/train/tensor_act_model_layers_58_mlp_waleed_W_u/max_abs":2.71875,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/norm":0.025818231819103803,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/max_abs":0.240234375,"train/train/layer_model_layers_53/act/std":0.8294695115033349,"train/train/tensor_act_model_layers_54_input_layernorm/mean":0.0013628005981445312,"train/train/tensor_act_model_layers_56_post_attention_layernorm/std":1.00000104868504,"train/train/tensor_act_model_layers_21_self_attn/mean":-0.0009427070617675781,"train/train/tensor_param_model_layers_84_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp/max_abs":0.546875,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/std":6.974838407816369e-05,"train/train/tensor_act_model_layers_26_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/mean":-2.2223684936761856e-07,"train/train/tensor_param_model_layers_65_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/mean":0.0003299713134765625,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/max_abs":0.00019168853759765625,"train/train/tensor_grad_model_layers_35_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_75_self_attn_k_proj/max_abs":5.84375,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_15/act/std":0.9127672290504293,"train/train/tensor_param_model_layers_19_mlp_waleed_W_u_weight/std":0.0233154296875,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/std":0.03662109375,"train/train/tensor_act_model_layers_37_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp/norm":3143.7238071516185,"train/train/layer__model_layers_25/param/norm":19.57336708945985,"train/train/layer_model_layers_39/grad/mean":-4.272496795598505e-08,"train/train/tensor_act_model/norm":5792.6090087940365,"train/train/tensor_act_model_layers_31_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_o_proj/max_abs":0.8515625,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/norm":0.0057954680709651285,"train/train/tensor_act_model_layers_33_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56/mean":0.005870819091796875,"train/train/layer_model_layers_38/act/std":0.8711749139941893,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/norm":4.53125,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/max_abs":0.000743865966796875,"train/train/tensor_act_model_layers_29_post_attention_layernorm/std":1.0000002867439706,"train/train/tensor_param_model_layers_23_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/norm":3.171875,"train/train/layer_model_layers_22/act/norm":20320.756533983054,"train/train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/std":3.325857791060527e-05,"train/train/tensor_act_model_layers_8/max_abs":26.625,"train/train/tensor_grad_model_layers_25_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_waleed_W_g/std":0.4238281565190449,"train/train/tensor_act_model_layers_76_post_attention_layernorm/std":1.0000006298940043,"train/train/tensor_act_model_layers_40_self_attn_o_proj/std":0.051453281964750505,"train/train/tensor_grad_model_layers_65_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_mlp_waleed_W_g_weight/std":0.0230712890625,"train/train/tensor_act_model_layers_5_self_attn_k_proj/std":1.6503946767001105,"train/train/tensor_act_model_layers_54_mlp_waleed/mean":0.0018768310546875,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_53/grad/max_abs":0.0013580322265625,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/mean":-0.00014019012451171875,"train/train/layer__model_layers_2/param/mean":0.0015452201950382702,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/std":0.052490234375,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/std":2.9294470883656555e-05,"train/train/tensor_act_model_layers_89_self_attn_q_proj/mean":-0.1024169921875,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_mlp_waleed_W_g_weight/max_abs":0.1513671875,"train/train/tensor_act_model_layers_91_self_attn/std":0.6250062124844148,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/norm":6.875,"train/train/tensor_param_model_layers_79_mlp_waleed_W_u_weight/mean":0.000202178955078125,"train/train/tensor_grad_model_layers_58_mlp_waleed_W_u_weight/norm":0.016269161041553973,"train/train/tensor_act_model_layers_65_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_k_proj/std":1.2441458290266731,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/max_abs":0.13671875,"train/train/tensor_act_model_layers_38_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/max_abs":0.000820159912109375,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/std":0.044677734375,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/norm":3.75,"train/train/tensor_act_model_layers_86_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/max_abs":0.00020503997802734375,"train/train/tensor_act_model_layers_4_self_attn/max_abs":0.76953125,"train/train/tensor_grad_model_layers_37_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/std":0.00013690307790463572,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/norm":4.59375,"train/train/tensor_param_model_layers_79_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_17_mlp_waleed_W_u_weight/mean":0.00013637542724609375,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/norm":0.02146051989837497,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_49/param/std":0.04826388807213495,"train/train/tensor_act_model_layers_66_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn_k_proj/norm":7166.776426816788,"train/train/tensor_param_model_layers_74_mlp_waleed_W_u_weight/norm":6.375,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/norm":3.546875,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/std":0.00015233931314953254,"train/train/layer_model_layers_61/act/frac_near_user_limit":0,"train/train/layer_model_layers_15/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/max_abs":0.00037384033203125,"train/train/tensor_param_model_layers_85_mlp_waleed_W_g_weight/mean":0.0003566741943359375,"train/train/tensor_act_model_layers_23_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/std":0.037841796875,"train/train/tensor_act_model_layers_22_self_attn_k_proj/std":0.9707051896931049,"train/train/tensor_param_model_layers_62_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_post_attention_layernorm/std":1.000000091269608,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp_down_proj/norm":347.0232181655934,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_waleed_W_u/mean":0.00753021240234375,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/max_abs":0.00092315673828125,"train/train/tensor_param_model_layers_73_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_mlp_waleed_W_g_weight/mean":8.323695510625839e-08,"train/train/layer_model_layers_11/grad/norm":0.04194500132592581,"train/train/tensor_act_model_layers_85_mlp/norm":2422.5119902145016,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model/max_abs":6.5625,"train/train/tensor_act_model_layers_30_self_attn_v_proj/std":0.4482457889565954,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn_v_proj/max_abs":2.390625,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/norm":0.0018181104894324886,"train/train/tensor_act_model_layers_30_mlp_waleed_W_u/mean":0.00257110595703125,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/norm":0.012150672510749902,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/norm":0.02313248526081806,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/mean":2.2423046175390482e-07,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/norm":0.004879211051283241,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/mean":6.742775440216064e-07,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_post_attention_layernorm/mean":-0.0062255859375,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn/mean":0.0049285888671875,"train/train/tensor_act_model_layers_2_self_attn/mean":0.0013751983642578125,"train/train/tensor_act_model_layers_85_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_mlp_waleed_W_g_weight/max_abs":0.1875,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/norm":5.71875,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/mean":-5.784386303275824e-09,"train/train/tensor_act_model_layers_85_self_attn/std":0.4335939456615995,"train/train/tensor_act_model_layers_19_self_attn/std":0.0629892414231707,"train/train/tensor_act_model_layers_84_input_layernorm/std":1.0000003328313498,"train/train/tensor_act_model_layers_13_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/norm":0.004677484660959401,"train/train/tensor_act_model_layers_5_self_attn_q_proj/std":1.3359378388053063,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_80_mlp/max_abs":2.765625,"train/train/tensor_act_model_layers_18_input_layernorm/max_abs":5.21875,"train/train/tensor_param_model_layers_31_mlp_waleed_W_u_weight/std":0.0245361328125,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/norm":0.008147393233772273,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/std":0.031494140625,"train/train/tensor_act_model_layers_39_self_attn_k_proj/std":0.934572068873462,"train/train/tensor_act_model_layers_92_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_v_proj/std":0.29882820341550537,"train/train/tensor_param_model_layers_82_mlp_waleed_W_u_weight/norm":7.53125,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/max_abs":0.00095367431640625,"train/train/layer_model_layers_74/act/mean":-6.195902824401855e-05,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_post_attention_layernorm/mean":0.00130462646484375,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_15_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/std":0.02587890625,"train/train/tensor_param_model_layers_59_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/max_abs":0.0020599365234375,"train/train/layer_model_layers_69/grad/std":5.983432150677956e-05,"train/train/tensor_act_model_layers_31_self_attn/max_abs":1.90625,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/max_abs":0.00013256072998046875,"train/train/tensor_act_model_layers_53_self_attn_q_proj/norm":5284.36218176666,"train/train/tensor_param_model_layers_83_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/max_abs":0.0001316070556640625,"train/train/tensor_param_model_layers_23_mlp_waleed_W_u_weight/norm":4.1875,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_q_proj/norm":5619.114351857181,"train/train/tensor_act_model_layers_25_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_mlp_waleed_W_u_weight/mean":-0.00014019012451171875,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/std":0.000133543391891383,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/std":0.033203125,"train/train/tensor_act_model_layers_66_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_58_mlp_waleed_W_g/mean":-0.0069122314453125,"train/train/tensor_act_model_layers_37_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_67/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn_o_proj/std":0.1948249239022955,"train/train/tensor_param_model_layers_54_mlp_waleed_W_g_weight/norm":5.03125,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/norm":0.014802063581098692,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_post_attention_layernorm/mean":0.0019989013671875,"train/train/tensor_act_model_layers_23_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/norm":2.90625,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/mean":1.2814998626708984e-05,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/std":0.0537109375,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/mean":1.346692442893982e-06,"train/train/tensor_act_model_layers_59_mlp/max_abs":0.6328125,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/std":2.2180060231872913e-05,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/mean":0.00019931793212890625,"train/train/tensor_grad_model_layers_20_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/norm":0.005815280460777992,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_waleed_W_g_weight/std":4.215047674523378e-05,"train/train/tensor_act_model_layers_59_self_attn/max_abs":1.4453125,"train/train/layer_model_layers_49/grad/max_abs":0.0007476806640625,"train/train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_waleed_W_g_weight/norm":5.53125,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp/std":2.3281250358027896,"train/train/tensor_param_model_layers_58_mlp_waleed_W_g_weight/max_abs":0.1552734375,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/norm":6.125,"train/train/tensor_act_model_layers_10_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83/std":3.332038854789329,"train/train/layer_model_layers_86/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_mlp_waleed_W_g_weight/std":0.0233154296875,"train/train/tensor_grad_model_layers_23_mlp_waleed_W_g_weight/mean":8.731149137020111e-08,"train/train/tensor_act_model_layers_50/norm":15206.469642457607,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/max_abs":0.0791015625,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/mean":7.30506144464016e-08,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/mean":-4.0046870708465576e-08,"train/train/tensor_grad_model_layers_29_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_mlp_waleed_W_u/std":0.2846692755507218,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/max_abs":0.208984375,"train/train/tensor_act_model_layers_86_self_attn_o_proj/mean":-0.00782012939453125,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/std":0.00010636110926220804,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/norm":3.53125,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_mlp_waleed_W_u/std":0.34472797091367846,"train/train/tensor_act_model_layers_2_self_attn_v_proj/max_abs":2.265625,"train/train/tensor_param_model_layers_52_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_43_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_mlp_waleed_W_g_weight/norm":5.71875,"train/train/tensor_act_model_layers_91_mlp_waleed/mean":-0.017120361328125,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp/std":0.5947418028404788,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_q_proj/mean":0.063232421875,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/norm":7.375,"train/train/tensor_act_model_layers_4_mlp_waleed_W_g/norm":3118.7239476239333,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/mean":-5.5530108511447906e-08,"train/train/tensor_act_model_layers_47_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_17/param/std":0.048257782168919065,"train/train/tensor_act_model_layers_6_self_attn_q_proj/norm":9900.21311940244,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/max_abs":0.0016021728515625,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/norm":7.03125,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/mean":5.960464477539062e-07,"train/train/tensor_grad_model_layers_65_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/std":1.069446862179389e-05,"train/train/tensor_act_model_layers_48_self_attn_k_proj/mean":-0.00789642333984375,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/mean":-6.370246410369873e-07,"train/train/layer_model_layers_39/act/norm":19679.255184314083,"train/train/tensor_act_model_layers_29/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn/frac_near_dtype_limit":0,"train/train/layer__model_layers_25/param/max_abs":1,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/max_abs":0.2490234375,"train/train/tensor_act_model_layers_87_self_attn_o_proj/std":0.6035442013749587,"train/train/tensor_act_model_layers_82_self_attn/std":0.3994216229827387,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/std":5.485545511507912e-05,"train/train/tensor_act_model_layers_86_self_attn/norm":2046.6089941697269,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_mlp_waleed_W_u_weight/norm":0.01916892625785517,"train/train/tensor_act_model_layers_80_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_36/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_input_layernorm_weight/std":0.00102996826171875,"train/train/tensor_act_model_layers_15_mlp_waleed_W_g/norm":2084.4588751489378,"train/train/tensor_param_model_norm_weight/norm":11.3125,"train/train/tensor_act_model_layers_31_mlp_down_proj/std":0.06091370545236591,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/max_abs":0.1640625,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/std":3.9341614356124304e-05,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/std":3.481804290341925e-05,"train/train/tensor_grad_model_layers_62_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_mlp/norm":469.63469681359896,"train/train/tensor_act_model_layers_33_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_mlp_waleed_W_u_weight/mean":9.266659617424011e-08,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/std":2.7033653879270598e-05,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/mean":4.895031452178955e-06,"train/train/layer_model_layers_19/grad/std":4.8324078510171895e-05,"train/train/layer_model_layers_68/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_42/act/frac_near_dtype_limit":0,"train/train/layer__model_layers_71/param/std":0.05359661120831072,"train/train/tensor_act_model_layers_71_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn_q_proj/norm":4929.934126971062,"train/train/layer__model_layers_65/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/mean":-1.6558915376663208e-06,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/std":7.393667587214814e-05,"train/train/tensor_act_model_layers_81_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/std":6.163996632922912e-05,"train/train/tensor_act_model_layers_78_mlp_down_proj/std":0.20166084274857496,"train/train/tensor_act_model_layers_43_self_attn_v_proj/norm":2221.8543496142834,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/max_abs":0.16796875,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_self_attn_q_proj/norm":4777.393709480815,"train/train/tensor_act_model_layers_19_mlp_waleed_W_u/norm":2096.836884777497,"train/train/tensor_act_model_layers_29_self_attn_o_proj/mean":0.0007238388061523438,"train/train/tensor_act_model_layers_40_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/norm":0.0075802862880385855,"train/train/tensor_param_model_layers_9_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn_v_proj/max_abs":3.875,"train/train/layer__model_layers_5/param/max_abs":1,"train/train/layer_model_layers_82/act/mean":0.019277412444353104,"train/train/tensor_grad_model_layers_71_mlp_waleed_W_u_weight/norm":0.020996803449124746,"train/train/layer__model_layers_27/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_input_layernorm/max_abs":5.59375,"train/train/tensor_act_model_layers_12_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn_o_proj/mean":-0.0014476776123046875,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/norm":0.0003808600799388964,"train/train/tensor_act_model_layers_76_self_attn/max_abs":2.359375,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/mean":4.009343683719635e-07,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/std":0.025634765625,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/std":0.05224609375,"train/train/tensor_act_model_layers_47_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/max_abs":0.2353515625,"train/train/tensor_act_model_layers_61_self_attn_k_proj/norm":6535.994756826891,"train/train/tensor_act_model_layers_54_post_attention_layernorm/max_abs":5.03125,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/mean":2.0616425899788737e-08,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/max_abs":0.000308990478515625,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/mean":5.1975250244140625e-05,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/mean":0.00011587142944335938,"train/train/global/act/std":1.0446034540204123,"train/train/layer_model_layers_45/grad/max_abs":0.00136566162109375,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_waleed_W_u_weight/max_abs":0.10205078125,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/std":0.0240478515625,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/mean":-0.0002689361572265625,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/std":4.27936653471187e-05,"train/train/tensor_act_model_layers_72_self_attn_v_proj/norm":2771.6031339325955,"train/train/layer__model_layers_24/param/max_abs":1,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/norm":5.25,"train/train/tensor_act_model_layers_5_self_attn_v_proj/max_abs":2.453125,"train/train/tensor_grad_model_layers_44_mlp_waleed_W_g_weight/std":3.886557291060676e-05,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/mean":-1.2922100722789764e-08,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_v_proj/norm":2805.884040221474,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/norm":0.006130730038506358,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_waleed/max_abs":2.953125,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/norm":0.008378895490590548,"train/train/layer__model_layers_43/param/mean":0.0014673819222055992,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp/std":0.046875327021018556,"train/train/tensor_act_model_layers_18_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/max_abs":0.000843048095703125,"train/train/layer_model_layers_67/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/std":5.5026906512801745e-05,"train/train/tensor_act_model_layers_9/mean":-0.006622314453125,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/std":0.037841796875,"train/train/tensor_act_model_layers_21_mlp_waleed_W_g/std":0.2851562811546521,"train/train/tensor_act_model_layers_63_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/norm":4.875,"train/train/tensor_act_model_layers_22_post_attention_layernorm/max_abs":5.1875,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_o_proj/norm":394.77467145618203,"train/train/tensor_act_model_layers_87_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/max_abs":9.775161743164062e-05,"train/train/tensor_act_model_layers_27_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_47/act/norm":18896.041233388984,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/max_abs":0.0002193450927734375,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/max_abs":0.2333984375,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/max_abs":0.1923828125,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/std":2.6589184914496386e-05,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/std":0.0250244140625,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/max_abs":0.1259765625,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp_waleed_W_g/std":0.6054688853602104,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/max_abs":0.000179290771484375,"train/train/tensor_act_model_layers_47_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn_k_proj/std":1.0410214078645956,"train/train/layer_model_layers_76/act/std":0.9347159012070481,"train/train/tensor_act_model_layers_66_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/norm":0.013932235367087818,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/mean":-1.130392774939537e-07,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/max_abs":0.0009307861328125,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/max_abs":0.000270843505859375,"train/train/tensor_act_model_layers_74_mlp_waleed_W_u/norm":3826.135835185384,"train/train/tensor_act_model_layers_72_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_73/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_down_proj/mean":0.0008726119995117188,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/max_abs":0.00103759765625,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_mlp_waleed_W_g_weight/max_abs":0.000946044921875,"train/train/layer_model_layers_52/grad/mean":5.437558089701136e-09,"train/train/tensor_param_model_layers_9_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/max_abs":0.00029754638671875,"train/train/tensor_act_model_layers_42_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn_o_proj/std":0.2670985080762469,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/std":0.037109375,"train/train/tensor_param_model_layers_44_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/mean":1.869630068540573e-07,"train/train/tensor_param_model_layers_12_mlp_waleed_W_u_weight/mean":0.00015544891357421875,"train/train/tensor_act_model_layers_45/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/norm":1.146981698459728,"train/train/tensor_param_model_layers_45_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_down_proj/std":0.05969250884754629,"train/train/tensor_param_model_layers_4_mlp_waleed_W_g_weight/max_abs":0.11083984375,"train/train/tensor_act_model_layers_34_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/norm":2.78125,"train/train/tensor_param_model_layers_5_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/std":0.036376953125,"train/train/tensor_act_model_layers_54_self_attn_v_proj/norm":2207.1736176465424,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/max_abs":0.1044921875,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/max_abs":0.1748046875,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/norm":6.46875,"train/train/tensor_act_model_layers_4_input_layernorm/mean":-0.0113983154296875,"train/train/tensor_act_model_layers_20_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/std":7.820725199975728e-05,"train/train/tensor_act_model_layers_3_mlp_down_proj/mean":-0.0103302001953125,"train/train/tensor_param_model_layers_81_mlp_waleed_W_u_weight/std":0.040283203125,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/norm":7.9375,"train/train/tensor_act_model_layers_82_mlp_waleed_W_u/norm":4877.3530470031155,"train/train/layer_model_layers_33/grad/mean":-6.743007907629385e-09,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_post_attention_layernorm/mean":0.002483367919921875,"train/train/tensor_act_model_layers_66_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/mean":0.00016689300537109375,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/std":4.232082578321409e-05,"train/train/tensor_act_model_layers_66_self_attn_k_proj/max_abs":4.6875,"train/train/tensor_act_model_layers_42_mlp_waleed/norm":836.8474333867426,"train/train/layer__model_layers_62/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_mlp_waleed_W_g_weight/mean":8.94651748239994e-08,"train/train/tensor_act_model_layers_31_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/max_abs":0.00012302398681640625,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/mean":4.0531158447265625e-05,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/std":0.0002687034802269767,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/std":9.716616303463017e-05,"train/train/tensor_act_model_layers_57_self_attn_v_proj/std":0.512698170790198,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/mean":1.8596649169921875e-05,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/mean":0.000766754150390625,"train/train/layer__model_layers_31/param/norm":19.70150791341871,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp/max_abs":3.28125,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_input_layernorm/std":1.0000009940408352,"train/train/layer_model_layers_19/act/norm":20437.720363104396,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/norm":4.21875,"train/train/tensor_act_model_layers_37_mlp_waleed/norm":808.7878947171113,"train/train/tensor_act_model_layers_6_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/std":5.632923736727226e-06,"train/train/tensor_param_model_layers_3_mlp_waleed_W_u_weight/mean":0.00013637542724609375,"train/train/tensor_act_model_layers_54_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_q_proj/std":0.8955095321281004,"train/train/layer_model_layers_31/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/mean":1.30385160446167e-08,"train/train/tensor_act_model_layers_78_mlp_waleed_W_g/std":0.4980470062909,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/norm":9.4375,"train/train/tensor_param_model_layers_83_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/mean":-2.232263796031475e-08,"train/train/layer__model_layers_0/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_mlp_waleed_W_u_weight/mean":-0.000263214111328125,"train/train/tensor_act_model_layers_8_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/norm":0.004332690096837863,"train/train/tensor_act_model_layers_84_self_attn_o_proj/norm":5550.08046715998,"train/train/tensor_act_model_norm/mean":-0.0088348388671875,"train/train/tensor_act_model_layers_53_self_attn_k_proj/std":0.7783226936906983,"train/train/tensor_act_model_layers_47_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/max_abs":0.00019550323486328125,"train/train/tensor_act_model_layers_0_mlp_waleed/max_abs":20.75,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/std":5.3189096498590106e-05,"train/train/tensor_grad_model_layers_35_mlp_waleed_W_g_weight/norm":0.01309307245490802,"train/train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_56/grad/std":4.46309371396318e-05,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/mean":-3.348104655742645e-07,"train/train/tensor_act_model_layers_89_mlp/std":0.585937650998414,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/max_abs":0.16015625,"train/train/tensor_grad_model_layers_0_mlp_waleed_W_g_weight/std":0.0017662131176034064,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn/mean":0.0002872943878173828,"train/train/tensor_act_model_layers_66_self_attn/max_abs":3.65625,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp_down_proj/norm":637.9534793566776,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/norm":0.001070294541122137,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/max_abs":0.00014495849609375,"train/train/tensor_act_model_layers_40_post_attention_layernorm/std":1.0000011476993458,"train/train/tensor_act_model_layers_47_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/std":0.037109375,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_36/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_64/param/mean":0.0016377132136066693,"train/train/tensor_param_model_layers_53_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/norm":3.828125,"train/train/tensor_act_model_layers_78_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_o_proj/max_abs":3,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/mean":-2.1117739379405975e-07,"train/train/tensor_act_model_layers_69_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_down_proj/norm":6043.683450899033,"train/train/tensor_act_model_layers_13_self_attn_o_proj/std":0.08130071005163605,"train/train/tensor_act_model_layers_91_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/std":0.000105352817836072,"train/train/tensor_act_model_layers_93_post_attention_layernorm/std":1.0000000700820213,"train/train/tensor_grad_model_layers_33_mlp_waleed_W_u_weight/std":3.4571957968627836e-05,"train/train/layer_model_layers_22/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/mean":-1.0570511221885681e-07,"train/train/tensor_act_model_layers_86_self_attn_k_proj/max_abs":6.625,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/max_abs":0.00124359130859375,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/max_abs":0.00023365020751953125,"train/train/tensor_grad_model_layers_66_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_waleed_W_g/max_abs":2.421875,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_o_proj/max_abs":1.7578125,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_7/act/norm":23518.190342484744,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/norm":4.25,"train/train/tensor_act_model_layers_8_post_attention_layernorm/std":1.0000000744112159,"train/train/tensor_act_model_layers_55_self_attn_o_proj/max_abs":0.7734375,"train/train/tensor_act_model_layers_55_mlp_waleed/std":0.13549874391529537,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/std":0.0001179406838434324,"train/train/tensor_act_model_layers_65_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/mean":-9.059906005859375e-06,"train/train/tensor_act_model_layers_23_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp/mean":0.00251007080078125,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/norm":3.421875,"train/train/tensor_grad_model_layers_67_mlp_waleed_W_u_weight/norm":0.020146964544065022,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/max_abs":0.001617431640625,"train/train/layer__model_layers_1/param/mean":0.0015930675679175605,"train/train/tensor_act_model_layers_71_post_attention_layernorm/max_abs":5.15625,"train/train/layer_model_layers_36/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_v_proj/std":0.5205107415064981,"train/train/tensor_act_model_layers_85_mlp_waleed_W_g/std":0.6523439989474958,"train/train/layer_model_layers_77/grad/norm":0.06193114945193662,"train/train/tensor_act_model_layers_68/mean":0.0140838623046875,"train/train/tensor_act_model_layers_54_self_attn_k_proj/max_abs":4.8125,"train/train/tensor_grad_model_layers_14_mlp_waleed_W_g_weight/std":9.992100566160616e-05,"train/train/tensor_act_model_layers_5_post_attention_layernorm/norm":5792.612304691071,"train/train/tensor_act_model_layers_51_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp_waleed/mean":-0.00040340423583984375,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70/norm":15751.332854823413,"train/train/tensor_act_model_layers_69/max_abs":25.375,"train/train/tensor_act_model_layers_34_mlp_down_proj/max_abs":0.5,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/mean":9.03010368347168e-06,"train/train/tensor_grad_model_layers_92_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_input_layernorm/std":1.0000000832005718,"train/train/tensor_act_model_layers_76_self_attn_v_proj/norm":2691.773625600188,"train/train/tensor_act_model_layers_7_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/mean":-1.6600824892520905e-07,"train/train/tensor_param_model_layers_10_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_80_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/max_abs":0.00038909912109375,"train/train/tensor_grad_model_layers_8_mlp_waleed_W_g_weight/mean":-8.434290066361427e-08,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_mlp_waleed_W_u/max_abs":2.09375,"train/train/tensor_act_model_layers_76_self_attn_q_proj/std":1.054687526049437,"train/train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_77_self_attn/norm":2034.2776746576142,"train/train/tensor_act_model_layers_89_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/max_abs":0.000736236572265625,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/max_abs":0.228515625,"train/train/tensor_act_model_layers_50_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_mlp_waleed_W_u_weight/std":8.728143645981315e-05,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/mean":6.225891411304474e-07,"train/train/tensor_act_model_layers_83_self_attn/mean":-0.016357421875,"train/train/tensor_act_model_layers_8_self_attn_k_proj/max_abs":8.375,"train/train/tensor_grad_model_layers_48_mlp_waleed_W_u_weight/norm":0.01525933836901891,"train/train/tensor_act_model_layers_38_self_attn_o_proj/norm":1374.518252251199,"train/train/tensor_act_model_layers_25_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_k_proj/max_abs":5.125,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/norm":3.5,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/mean":7.49350874684751e-08,"train/train/tensor_grad_model_layers_23_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp/norm":476.7418686017982,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/mean":-0.000209808349609375,"train/train/tensor_act_model_layers_59_mlp/norm":476.3652992310078,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/std":0.025634765625,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_68/grad/norm":0.04635807238249556,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_k_proj/std":0.8662127674345795,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/std":0.04736328125,"train/train/tensor_act_model_layers_69_mlp/max_abs":0.97265625,"train/train/tensor_grad_model_layers_23_mlp_waleed_W_u_weight/std":3.4158258584264475e-05,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/max_abs":0.00121307373046875,"train/train/tensor_act_model_layers_79_mlp_down_proj/mean":-0.0009250640869140625,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/mean":-8.539063856005669e-08,"train/train/tensor_param_model_layers_37_mlp_waleed_W_u_weight/std":0.025634765625,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/mean":1.6222475096583366e-07,"train/train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/max_abs":0.0004405975341796875,"train/train/layer_model_layers_77/grad/max_abs":0.00150299072265625,"train/train/tensor_act_model_layers_45_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_post_attention_layernorm/norm":5792.609619142986,"train/train/tensor_param_model_layers_73_mlp_waleed_W_g_weight/max_abs":0.1884765625,"train/train/tensor_act_model_layers_68/std":2.6875246260919603,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_mlp_waleed_W_g_weight/mean":7.35744833946228e-07,"train/train/tensor_act_model_layers_70_mlp/norm":741.2930685159407,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/std":2.8688410218867762e-05,"train/train/tensor_act_model_layers_41_input_layernorm/norm":5792.608642584983,"train/train/tensor_act_model_layers_1_mlp_waleed_W_u/norm":7542.810677184555,"train/train/layer_model_layers_48/act/mean":0.0002608299255371094,"train/train/tensor_param_model_layers_81_mlp_waleed_W_g_weight/mean":-0.00011682510375976562,"train/train/tensor_act_model_layers_81_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93/max_abs":46.5,"train/train/tensor_grad_model_layers_93_mlp_waleed_W_g_weight/std":0.0001299511144333184,"train/train/tensor_act_model_layers_26/max_abs":26.375,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_mlp_waleed_W_g_weight/mean":-1.1670636013150215e-07,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/max_abs":0.232421875,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/std":4.7553644091441115e-05,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/norm":6.03125,"train/train/layer_model_layers_35/act/norm":19794.15367030751,"train/train/tensor_act_model_layers_43_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/mean":1.511070877313614e-07,"train/train/tensor_act_model_layers_73_mlp_waleed/norm":1694.2736408632336,"train/train/tensor_act_model_layers_60_mlp/mean":0.000965118408203125,"train/train/tensor_act_model_layers_89_mlp_down_proj/std":0.585937650998414,"train/train/tensor_act_model_layers_39_mlp_down_proj/mean":0.00243377685546875,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/mean":-3.475695848464966e-06,"train/train/tensor_act_model_layers_23_self_attn_k_proj/mean":0.045166015625,"train/train/tensor_act_model_layers_74_mlp/max_abs":1.296875,"train/train/tensor_act_model_layers_78_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/mean":0.00011110305786132812,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/mean":-1.9937753677368164e-05,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_85_self_attn_o_proj/mean":-0.00763702392578125,"train/train/tensor_act_model_layers_49_mlp_down_proj/max_abs":1.0390625,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/norm":0.9512540871811833,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/mean":-0.00023555755615234375,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_54/act/mean":0.005621761083602905,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn/mean":-0.00043392181396484375,"train/train/tensor_act_model_layers_65_input_layernorm/norm":5792.612548835219,"train/train/tensor_act_model_layers_18_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_20_mlp_waleed_W_g/max_abs":2.15625,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp_waleed_W_u/mean":0.0123291015625,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_77/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_mlp_down_proj/max_abs":1.453125,"train/train/tensor_act_model_layers_45_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model/std":1.000000045984051,"train/train/layer_model_layers_8/act/norm":26071.019841260157,"train/train/tensor_param_model_layers_25_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/max_abs":0.0004291534423828125,"train/train/tensor_act_model_layers_65_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_post_attention_layernorm/max_abs":5.4375,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/std":0.0230712890625,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/std":1.375115640315416e-05,"train/train/tensor_act_model_layers_64_self_attn_v_proj/mean":-0.0035858154296875,"train/train/tensor_act_model_layers_92_self_attn_k_proj/max_abs":5.90625,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/max_abs":0.2578125,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/std":6.993453097707889e-05,"train/train/tensor_act_model_layers_85_input_layernorm/mean":0.008087158203125,"train/train/tensor_act_model_layers_89_self_attn_o_proj/max_abs":7.03125,"train/train/tensor_act_model_layers_55_mlp_waleed_W_g/norm":3013.908533554625,"train/train/tensor_act_model_layers_29_self_attn_v_proj/mean":-0.00409698486328125,"train/train/tensor_param_model_layers_2_mlp_waleed_W_g_weight/max_abs":0.1298828125,"train/train/tensor_act_model_layers_6_self_attn_v_proj/std":0.40429690123337647,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/norm":7.53125,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/norm":5.875,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/mean":0.000293731689453125,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/max_abs":0.00136566162109375,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/std":0.0306396484375,"train/train/tensor_grad_model_layers_73_mlp_waleed_W_u_weight/max_abs":0.000629425048828125,"train/train/tensor_act_model_layers_48_mlp_down_proj/norm":587.8200903843738,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/mean":-1.210719347000122e-07,"train/train/tensor_act_model_layers_71_self_attn_o_proj/norm":791.4331879325297,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_mlp_waleed_W_u_weight/max_abs":0.12890625,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_87/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/mean":-0.0004482269287109375,"train/train/layer_model_layers_74/act/std":0.9496593752041682,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/std":0.00017004326626533553,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/max_abs":0.000751495361328125,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/norm":2.828125,"train/train/tensor_grad_model_layers_71_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_mlp_waleed_W_u/norm":7986.3101128682065,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/max_abs":0.00054931640625,"train/train/tensor_param_model_layers_53_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12/max_abs":26.25,"train/train/layer_model_layers_93/act/max_abs":46.5,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/std":0.0264892578125,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/std":8.418442817535595e-05,"train/train/tensor_param_model_layers_50_mlp_waleed_W_u_weight/norm":4.9375,"train/train/tensor_act_model_layers_33_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/layer__model_layers_75/param/mean":0.0013674201905820374,"train/train/tensor_param_model_layers_35_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_o_proj/std":0.17260802730337652,"train/train/layer_model_layers_31/act/std":0.8534756968611088,"train/train/tensor_param_model_layers_76_mlp_waleed_W_g_weight/mean":4.798173904418945e-06,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/mean":1.6379635781049728e-07,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/std":1.8834790981780088e-05,"train/train/tensor_act_model_layers_51_self_attn_k_proj/std":1.0078125295712963,"train/train/layer_model_layers_10/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_q_proj/norm":6129.111472824354,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/max_abs":0.000499725341796875,"train/train/tensor_act_model_layers_25_post_attention_layernorm/norm":5792.439086915722,"train/train/layer_model_layers_63/grad/norm":0.04426235405733313,"train/train/tensor_act_model_layers_80_self_attn_o_proj/max_abs":5.5625,"train/train/tensor_act_model_layers_79_self_attn/mean":0.0008535385131835938,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_waleed_W_g/norm":3229.072296073595,"train/train/layer_model_layers_34/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/max_abs":0.0002899169921875,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/std":0.036376953125,"train/train/tensor_act_model_layers_16_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/std":0.037841796875,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/mean":9.261071681976318e-06,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/norm":0.016144346866067014,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/std":8.923502722627703e-05,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/mean":2.1338462829589844e-05,"train/train/tensor_act_model_layers_51_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/std":5.640098267763468e-05,"train/train/tensor_act_model_layers_35_input_layernorm/max_abs":5.40625,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/norm":0.004970268013127584,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90/max_abs":35.5,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/mean":8.487701416015625e-05,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_73_self_attn/norm":1904.6041544610314,"train/train/tensor_act_model_layers_59_input_layernorm/std":1.0000009941840804,"train/train/layer__model_layers_36/param/max_abs":1,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/std":0.03662109375,"train/train/tensor_param_model_layers_72_mlp_waleed_W_g_weight/mean":5.5789947509765625e-05,"train/train/tensor_act_model_layers_84_mlp_waleed/norm":3487.1964985909153,"train/train/tensor_act_model_layers_21_mlp_waleed/std":0.10363792242771151,"train/train/tensor_act_model_layers_65_mlp_waleed_W_g/mean":-0.011962890625,"train/train/tensor_act_model_layers_65_self_attn_v_proj/norm":2164.20703881888,"train/train/tensor_act_model_layers_13_mlp_down_proj/max_abs":0.57421875,"train/train/tensor_act_model_layers_40_mlp_waleed_W_u/mean":0.0006856918334960938,"train/train/tensor_act_model_layers_78_mlp_waleed/mean":0.0004391670227050781,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/std":7.605488225333113e-05,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn_q_proj/max_abs":6.40625,"train/train/tensor_param_model_layers_17_mlp_waleed_W_g_weight/max_abs":0.09765625,"train/train/layer_model_layers_62/act/max_abs":24.5,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_mlp_waleed_W_u_weight/std":4.6195643776253485e-05,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/std":3.8029470424814565e-05,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/max_abs":0.0010833740234375,"train/train/tensor_act_model_layers_84/mean":0.03057861328125,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_28_mlp/std":0.059387333363634984,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/max_abs":0.1826171875,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/max_abs":0.1240234375,"train/train/tensor_grad_model_layers_30_mlp_waleed_W_u_weight/std":4.029806126973792e-05,"train/train/tensor_act_model_layers_88_self_attn/std":0.37604446692330556,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/norm":0.004928305414038452,"train/train/layer_model_layers_45/act/std":0.8506511723261204,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/mean":-0.00014495849609375,"train/train/layer_model_layers_90/act/mean":-0.014527201652526855,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/mean":0.00037384033203125,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/std":6.831231820015589e-05,"train/train/tensor_act_model_layers_66/std":2.6562749414465094,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_35/param/std":0.04862098994071966,"train/train/tensor_act_model_layers_87_mlp_waleed/norm":4102.497632575474,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_self_attn_q_proj/norm":6683.5424059122315,"train/train/tensor_grad_model_layers_86_mlp_waleed_W_g_weight/mean":7.683411240577698e-09,"train/train/layer__model_layers_48/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_q_proj/mean":-0.0109100341796875,"train/train/tensor_act_model_layers_44_self_attn_o_proj/max_abs":0.875,"train/train/tensor_act_model_layers_45_mlp_waleed_W_u/std":0.3105469674996472,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/norm":0.013525106929737121,"train/train/tensor_act_model_layers_86_self_attn_v_proj/mean":0.002246856689453125,"train/train/tensor_act_model_layers_27_input_layernorm/norm":5791.985595703357,"train/train/tensor_act_model_layers_80_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/norm":4.625,"train/train/tensor_act_model_layers_47/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/std":0.0001374521901637376,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/mean":0.0004405975341796875,"train/train/tensor_act_model_layers_93_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/std":0.0213623046875,"train/train/tensor_act_model_layers_33_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/max_abs":0.000820159912109375,"train/train/tensor_act_model_layers_48_self_attn_k_proj/norm":5766.656066256808,"train/train/tensor_act_model_layers_37_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_14/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/std":0.0262451171875,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/max_abs":0.0001888275146484375,"train/train/tensor_act_model_layers_14_mlp_waleed_W_g/std":0.24414063675841288,"train/train/tensor_act_model_layers_21_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/std":0.039794921875,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/std":0.037841796875,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/std":0.0308837890625,"train/train/tensor_act_model_layers_75_mlp_waleed/norm":1905.2055424072576,"train/train/tensor_param_model_layers_35_mlp_waleed_W_u_weight/norm":4.53125,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/std":6.019002035346689e-05,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/max_abs":0.0023956298828125,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/mean":0.00021076202392578125,"train/train/tensor_act_model_layers_77_mlp/std":0.18920950420771576,"train/train/tensor_act_model_layers_8_mlp/norm":591.0908098819986,"train/train/tensor_act_model_layers_4_mlp/std":0.24560693622856009,"train/train/tensor_act_model_layers_28_mlp_waleed_W_u/max_abs":2.96875,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/norm":0.00865738295722059,"train/train/tensor_act_model_layers_65_post_attention_layernorm/mean":0.004268646240234375,"train/train/tensor_grad_model_layers_33_mlp_waleed_W_u_weight/mean":-2.8405338525772095e-07,"train/train/tensor_act_model_layers_77/mean":0.021728515625,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/global/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/max_abs":0.0001087188720703125,"train/train/tensor_grad_model_layers_11_mlp_waleed_W_u_weight/max_abs":0.00104522705078125,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/max_abs":0.000919342041015625,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/max_abs":0.1728515625,"train/train/layer_model_layers_44/grad/norm":0.030953449525201085,"train/train/layer__model_layers_8/param/norm":19.92609024032625,"train/train/tensor_act_model_layers_59_self_attn_k_proj/max_abs":5.40625,"train/train/tensor_act_model_layers_32_mlp_down_proj/mean":-0.000606536865234375,"train/train/tensor_act_model_layers_48/norm":15432.106675834344,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_mlp_waleed_W_g_weight/std":4.4421274083686255e-05,"train/train/tensor_act_model_layers_92_mlp_down_proj/max_abs":12.1875,"train/train/layer_model_layers_27/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/max_abs":0.294921875,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/std":0.040283203125,"train/train/tensor_param_model_layers_66_mlp_waleed_W_g_weight/std":0.0322265625,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/max_abs":0.00011873245239257812,"train/train/tensor_param_model_layers_11_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/max_abs":0.00013446807861328125,"train/train/tensor_act_model_layers_50_mlp_waleed_W_u/std":0.34814559493631353,"train/train/tensor_act_model_layers_80_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp_waleed/std":0.09875511760221069,"train/train/tensor_param_model_layers_92_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_75_input_layernorm/max_abs":5.40625,"train/train/layer_model_layers_76/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_v_proj/norm":2212.6042972123046,"train/train/tensor_act_model_layers_27_self_attn_q_proj/max_abs":8.1875,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/norm":0.0014199182564113425,"train/train/tensor_act_model_layers_9_self_attn_o_proj/mean":-0.001117706298828125,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/mean":-3.1170202419161797e-08,"train/train/tensor_param_model_layers_46_mlp_waleed_W_u_weight/norm":4.875,"train/train/tensor_param_model_layers_48_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/std":0.0001545444736567329,"train/train/layer_model_layers_17/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_63/grad/mean":7.667937628378548e-08,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp_waleed_W_g/max_abs":6.0625,"train/train/tensor_act_model_layers_68_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/std":5.263471600787214e-05,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/max_abs":0.000415802001953125,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/max_abs":0.0001354217529296875,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/norm":0.02127449517172218,"train/train/tensor_param_model_layers_71_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_mlp_waleed_W_u_weight/mean":-6.914138793945312e-05,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_k_proj/mean":0.0147705078125,"train/train/layer__model_layers_56/param/frac_near_user_limit":0,"train/train/layer_model_layers_49/grad/norm":0.03061917454043186,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/mean":-2.384185791015625e-07,"train/train/layer_model_layers_89/grad/max_abs":0.00128936767578125,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/mean":6.897607818245888e-08,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/mean":-9.489059448242188e-05,"train/train/tensor_act_model_layers_22_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_11_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/norm":6.5,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/mean":-1.9762665033340454e-06,"train/train/tensor_act_model_layers_47_post_attention_layernorm/max_abs":5.03125,"train/train/tensor_grad_model_layers_48_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_input_layernorm/std":1.0000000164145602,"train/train/layer__model_layers_56/param/std":0.050482973804655974,"train/train/tensor_act_model_layers_7_mlp/mean":0.00299835205078125,"train/train/tensor_act_model_layers_73/frac_near_user_limit":0,"train/train/layer_model_layers_91/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_mlp_waleed_W_g_weight/mean":-0.00020885467529296875,"train/train/tensor_param_model_layers_70_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/max_abs":0.00020885467529296875,"train/train/tensor_param_model_layers_86_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_27_mlp_waleed_W_g_weight/std":0.0242919921875,"train/train/layer__model_layers_0/param/max_abs":1,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/std":0.02197265625,"train/train/tensor_act_model_layers_31_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/max_abs":0.2373046875,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/max_abs":0.00055694580078125,"train/train/layer__model_layers_6/param/std":0.048296232118243944,"train/train/tensor_act_model_layers_41_mlp_waleed_W_g/mean":0.0068206787109375,"train/train/layer__model_layers_8/param/std":0.04916329441594554,"train/train/tensor_grad_model_layers_53_mlp_waleed_W_g_weight/max_abs":0.00081634521484375,"train/train/tensor_act_model_layers_57_input_layernorm/max_abs":4.9375,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_mlp_waleed_W_g_weight/norm":0.0173044086718,"train/train/tensor_param_model_layers_7_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer_model_layers_92/grad/std":0.00011816964867621059,"train/train/tensor_param_model_layers_82_mlp_waleed_W_g_weight/mean":-0.0002918243408203125,"train/train/tensor_act_model_layers_22_self_attn_o_proj/norm":624.365495066896,"train/train/tensor_act_model_layers_87_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_mlp_waleed_W_u_weight/mean":0.00010347366333007812,"train/train/tensor_act_model_layers_60_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/max_abs":0.125,"train/train/tensor_act_model_layers_90_self_attn_v_proj/norm":3960.181116304765,"train/train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6/mean":-0.0032825469970703125,"train/train/tensor_param_model_layers_44_mlp_waleed_W_g_weight/std":0.0264892578125,"train/train/tensor_grad_model_layers_40_mlp_waleed_W_g_weight/std":4.1251614097783165e-05,"train/train/tensor_act_model_layers_84_mlp_waleed_W_u/max_abs":4.6875,"train/train/tensor_param_model_layers_23_mlp_waleed_W_g_weight/mean":0.00011730194091796875,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/norm":0.0009169448369570518,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_post_attention_layernorm/std":1.0000009915274704,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_21_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/std":9.65806087439623e-05,"train/train/tensor_act_model_layers_14_self_attn_k_proj/std":0.9179687880455172,"train/train/tensor_act_model_layers_25/mean":0.00653076171875,"train/train/tensor_act_model_layers_7_self_attn_o_proj/max_abs":0.80078125,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_79_mlp_waleed_W_u_weight/mean":1.3562384992837906e-07,"train/train/tensor_act_model_layers_28_self_attn_k_proj/std":1.0703125253145964,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/max_abs":9.34600830078125e-05,"train/train/tensor_act_model_layers_63_self_attn_v_proj/std":0.4296877082775435,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/max_abs":0.1884765625,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/max_abs":6.031990051269531e-05,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_o_proj/std":0.23877105521645042,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/norm":0.0005779927282027274,"train/train/tensor_act_model_layers_48_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp_down_proj/norm":422.4777264688982,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/std":0.00010433239735015632,"train/train/tensor_param_model_layers_7_mlp_waleed_W_u_weight/max_abs":0.1171875,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/std":0.040283203125,"train/train/tensor_param_model_layers_55_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_mlp_waleed/max_abs":4.15625,"train/train/layer_model_layers_87/act/mean":0.007562920451164246,"train/train/tensor_act_model_layers_92_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_v_proj/std":0.49023446381328734,"train/train/tensor_act_model_layers_23_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_waleed_W_g_weight/std":0.0233154296875,"train/train/layer_model_layers_4/act/mean":0.0011543035507202148,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_post_attention_layernorm/norm":5792.615600588576,"train/train/layer_model_layers_87/act/max_abs":32,"train/train/tensor_act_model_layers_38_mlp_waleed_W_u/norm":2314.1336255819338,"train/train/tensor_param_model_layers_43_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_v_proj/mean":-0.00257110595703125,"train/train/tensor_param_model_layers_8_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_35_self_attn_o_proj/mean":-0.00029662251472473145,"train/train/tensor_act_model_layers_19_self_attn_o_proj/norm":365.7470157363681,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn_v_proj/mean":-0.0006093978881835938,"train/train/tensor_act_model_layers_20_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_post_attention_layernorm/max_abs":5.1875,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/max_abs":0.0008087158203125,"train/train/layer_model_layers_18/grad/mean":-4.597837664768123e-08,"train/train/tensor_act_model_layers_85_mlp_waleed_W_u/mean":-0.019134521484375,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/mean":-7.286667823791504e-06,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/std":2.5543579266430364e-05,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_input_layernorm/mean":0.006622314453125,"train/train/tensor_act_model_layers_46_mlp_waleed_W_u/std":0.31982535915317034,"train/train/tensor_act_model_layers_52_mlp_waleed_W_g/norm":2869.013334860885,"train/train/tensor_param_model_layers_76_mlp_waleed_W_g_weight/std":0.037109375,"train/train/tensor_act_model_layers_89_mlp_waleed_W_g/norm":5975.341156328031,"train/train/tensor_grad_model_layers_35_mlp_waleed_W_g_weight/std":3.618296056516583e-05,"train/train/layer__model_layers_14/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp_waleed_W_u/max_abs":2.21875,"train/train/tensor_act_model_layers_68_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/std":0.039306640625,"train/train/tensor_act_model_layers_14_mlp_down_proj/norm":401.4751021816768,"train/train/tensor_act_model_layers_76_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_mlp_waleed_W_g_weight/std":0.0262451171875,"train/train/tensor_param_model_layers_15_mlp_waleed_W_u_weight/mean":-4.291534423828125e-05,"train/train/tensor_param_model_layers_74_mlp_waleed_W_g_weight/mean":0.00020694732666015625,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_43/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn/max_abs":2.734375,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/norm":0.0023140847842542792,"train/train/tensor_act_model_layers_30_self_attn_o_proj/norm":850.5285257797606,"train/train/tensor_act_model_layers_58_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp/std":0.06713912660271366,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/norm":10.125,"train/train/tensor_act_model_layers_38_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_v_proj/norm":1793.2820286998356,"train/train/tensor_act_model_layers_9_self_attn_q_proj/norm":6941.1996062935,"train/train/tensor_act_model_layers_85_input_layernorm/std":1.0000001550651965,"train/train/tensor_param_model_layers_85_mlp_waleed_W_u_weight/norm":8,"train/train/tensor_act_model_layers_74_self_attn_o_proj/std":0.3930698799637309,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/std":0.00012992722953380286,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_61_self_attn_k_proj/max_abs":6.25,"train/train/tensor_act_model_layers_24_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/norm":6.8125,"train/train/tensor_param_model_layers_57_mlp_waleed_W_u_weight/std":0.029541015625,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/norm":5.15625,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_down_proj/std":0.06750523494245766,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_63/param/max_abs":1,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp_waleed/norm":1580.6579356480581,"train/train/tensor_param_model_layers_58_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/mean":-4.417961463332176e-08,"train/train/tensor_act_model_layers_70_input_layernorm/std":1.000000916501029,"train/train/tensor_param_model_layers_3_mlp_waleed_W_g_weight/max_abs":0.1298828125,"train/train/tensor_grad_model_layers_86_mlp_waleed_W_u_weight/norm":0.03162445785600568,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/mean":2.126907929778099e-07,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/std":3.90474683461828e-05,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/norm":0.02493575603438918,"train/train/tensor_act_model_layers_67_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_33/grad/norm":0.024643807075001745,"train/train/tensor_grad_model_layers_68_mlp_waleed_W_u_weight/mean":9.953510016202927e-08,"train/train/tensor_act_model_layers_2_self_attn_k_proj/norm":5168.803271113212,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_mlp_waleed_W_g_weight/std":0.024169921875,"train/train/tensor_act_model_layers_57_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/std":0.029296875,"train/train/layer_model_layers_81/grad/norm":0.06729296078770197,"train/train/tensor_act_model_layers_12_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn/max_abs":1.4609375,"train/train/layer_model_layers_17/grad/mean":1.2039959535304916e-07,"train/train/tensor_act_model_layers_49_mlp/std":0.09741486417275942,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/std":0.057373046875,"train/train/tensor_act_model_layers_12_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_input_layernorm/std":1.0000000621294,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/std":0.031005859375,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/norm":0.005654426303100345,"train/train/layer_model_layers_14/grad/std":6.104881007732148e-05,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/mean":-2.3067696020007133e-07,"train/train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/mean":2.547167241573334e-07,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/std":4.2186896536314235e-05,"train/train/tensor_act_model_layers_14_input_layernorm/max_abs":5.1875,"train/train/tensor_act_model_layers_54_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/norm":0.0006552929333474231,"train/train/tensor_param_model_layers_81_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_o_proj/norm":3620.590754566589,"train/train/tensor_act_model_layers_37_self_attn/std":0.10876771754840547,"train/train/tensor_grad_model_layers_19_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_waleed_W_g/norm":5447.461006445205,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/std":0.05224609375,"train/train/tensor_act_model_layers_19_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_mlp_waleed_W_g/max_abs":2.40625,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/norm":0.015570126977897667,"train/train/tensor_act_model_layers_50_self_attn_o_proj/std":0.10974322754416757,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/mean":-2.9441434890031815e-07,"train/train/layer__model_layers_16/param/mean":0.0015663230288977183,"train/train/tensor_param_model_layers_15_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer_model_layers_1/act/mean":-0.009921621531248093,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn/std":0.2949640231345105,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/norm":0.000775653286707667,"train/train/layer_model_layers_60/grad/std":3.887359954595055e-05,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_mlp_waleed_W_u_weight/max_abs":0.0009918212890625,"train/train/tensor_act_model_layers_66_mlp_waleed_W_g/norm":3356.7593734656634,"train/train/tensor_act_model_layers_56_self_attn_v_proj/max_abs":2.90625,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/max_abs":0.193359375,"train/train/tensor_act_model_layers_21_post_attention_layernorm/norm":5792.248413090686,"train/train/tensor_act_model_layers_41_mlp/norm":378.593975088037,"train/train/layer__model_layers_60/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp_waleed_W_u/std":0.394531311375075,"train/train/tensor_act_model_layers_80_self_attn_v_proj/norm":3731.6520571582596,"train/train/layer_model_layers_50/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_mlp_waleed_W_u_weight/max_abs":0.12255859375,"train/train/tensor_act_model_layers_71_mlp_waleed/norm":1537.2690019604502,"train/train/tensor_act_model_layers_52_mlp_waleed/max_abs":2.609375,"train/train/tensor_grad_model_layers_72_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_mlp_waleed_W_g/max_abs":2.359375,"train/train/layer__model_layers_56/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_52/param/mean":0.0015485885548703199,"train/train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn_q_proj/max_abs":6.46875,"train/train/layer_model_layers_78/grad/mean":1.922360216027675e-07,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/mean":0.0003757476806640625,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/norm":0.03384942853000081,"train/train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/std":0.057373046875,"train/train/tensor_param_model_layers_29_mlp_waleed_W_g_weight/norm":4.375,"train/train/tensor_act_model_layers_18_mlp_waleed/std":0.09619141798844574,"train/train/tensor_act_model_layers_29_mlp_waleed_W_g/max_abs":2.09375,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/max_abs":0.00070953369140625,"train/train/tensor_act_model_layers_27/max_abs":26.5,"train/train/tensor_act_model_layers_42_input_layernorm/mean":0.004123687744140625,"train/train/layer_model_layers_77/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_post_attention_layernorm/std":1.0000000076834112,"train/train/tensor_grad_model_layers_78_mlp_waleed_W_g_weight/std":6.688629885506809e-05,"train/train/tensor_param_model_layers_0_mlp_waleed_W_g_weight/norm":5,"train/train/tensor_act_model_layers_36_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/layer_model_layers_22/act/max_abs":26.375,"train/train/tensor_act_model_layers_65_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/mean":-7.385388016700745e-07,"train/train/tensor_act_model_layers_15_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/max_abs":0.00189208984375,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/norm":2.90625,"train/train/tensor_grad_model_layers_46_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_79/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_waleed/mean":-0.000759124755859375,"train/train/tensor_act_model_layers_62_post_attention_layernorm/mean":0.004650115966796875,"train/train/tensor_grad_model_layers_26_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/max_abs":0.0005645751953125,"train/train/tensor_act_model_layers_39_self_attn_k_proj/mean":-0.014434814453125,"train/train/tensor_act_model_layers_84_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/max_abs":0.0014801025390625,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/max_abs":0.1220703125,"train/train/layer__model_layers_57/param/std":0.051874579697248915,"_timestamp":1.7862589199085171e+09,"train/train/tensor_param_model_layers_59_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_73_self_attn_k_proj/max_abs":5.34375,"train/train/layer__model_layers_32/param/max_abs":1,"train/train/tensor_act_model_layers_91_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_88/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84/max_abs":29,"train/train/tensor_param_model_layers_22_mlp_waleed_W_g_weight/std":0.024169921875,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/norm":0.010386159051833698,"train/train/tensor_act_model_layers_31_input_layernorm/max_abs":5.5625,"train/train/tensor_act_model_layers_26_post_attention_layernorm/mean":0.0017452239990234375,"train/train/tensor_act_model_layers_41_mlp_waleed/max_abs":2.265625,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/std":9.038508402338144e-05,"train/train/tensor_grad_model_layers_67_mlp_waleed_W_g_weight/max_abs":0.0009613037109375,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_waleed_W_g/max_abs":2.546875,"train/train/layer__model_layers_80/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp_waleed_W_g/mean":0.002925872802734375,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_post_attention_layernorm/norm":5792.612792970766,"train/train/tensor_act_model_layers_48_mlp_waleed_W_g/std":0.35009871217336985,"train/train/tensor_param_model_layers_9_input_layernorm_weight/mean":1,"train/train/layer__model_layers_20/param/max_abs":1,"train/train/layer__model_layers_71/param/norm":21.719070366342226,"train/train/tensor_param_model_layers_92_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/norm":0.01641993179221164,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/norm":4.96875,"train/train/tensor_act_model_layers_27_self_attn_o_proj/norm":1382.6337837232854,"train/train/tensor_act_model_layers_18_mlp_down_proj/std":0.06689453480364134,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/max_abs":0.0052490234375,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/max_abs":0.0003757476806640625,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/norm":0.022283600360874865,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/std":7.327178200753281e-05,"train/train/tensor_grad_model_layers_44_mlp_waleed_W_u_weight/norm":0.014450806354427274,"train/train/layer__model_layers_84/param/std":0.05990754610238325,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/norm":5.125,"train/train/tensor_act_model_layers_24_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/norm":0.004329652828172717,"train/train/tensor_param_model_layers_4_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_waleed/norm":638.4791323804153,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/norm":0.008114821426957323,"train/train/layer_model_layers_40/act/std":0.8168616717610733,"train/train/tensor_act_model_layers_15_mlp/norm":383.73028644989967,"train/train/tensor_act_model_layers_76_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_mlp_waleed_W_g_weight/norm":4.25,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_mlp_waleed_W_u_weight/std":7.498920360509552e-05,"train/train/tensor_param_model_layers_21_mlp_waleed_W_u_weight/std":0.0238037109375,"train/train/tensor_act_model_layers_79/mean":0.01953125,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/max_abs":1.2265625,"train/train/tensor_param_model_layers_82_mlp_waleed_W_g_weight/max_abs":0.2158203125,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/std":0.00024468680443183764,"train/train/tensor_grad_model_layers_91_mlp_waleed_W_u_weight/norm":0.039072416954062106,"train/train/layer_model_layers_71/grad/max_abs":0.00122833251953125,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/std":0.0235595703125,"train/train/tensor_act_model_layers_24_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_post_attention_layernorm/norm":5792.609863293299,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_waleed_W_g/mean":-0.0094757080078125,"train/train/tensor_act_model_layers_69_self_attn/max_abs":2.875,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/max_abs":0.000606536865234375,"train/train/tensor_act_model_layers_37_self_attn_q_proj/max_abs":5.71875,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/mean":-0.0002803802490234375,"train/train/tensor_act_model_layers_37_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_input_layernorm/norm":5792.617553712742,"train/train/tensor_act_model_layers_44_mlp_waleed/max_abs":2.59375,"train/train/tensor_act_model_layers_77_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_69/act/norm":20441.44107048811,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/std":0.050537109375,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/std":0.042236328125,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_75/grad/norm":0.0581632356285878,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18/std":2.839873408802764,"train/train/tensor_act_model_layers_70_self_attn_k_proj/mean":0.0966796875,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/mean":1.1420343071222305e-07,"train/train/tensor_act_model_layers_43_mlp_waleed_W_u/std":0.30322384066204516,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp/norm":571.0707887493845,"train/train/tensor_grad_model_layers_64_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_k_proj/norm":8288.168866944447,"train/train/tensor_param_model_layers_29_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/max_abs":0.00038909912109375,"train/train/tensor_act_model_layers_44_self_attn_o_proj/mean":-0.00019657611846923828,"train/train/tensor_grad_model_layers_48_mlp_waleed_W_g_weight/mean":-1.1837983038276434e-07,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/max_abs":0.0018463134765625,"train/train/tensor_act_model_layers_65/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_mlp_waleed_W_u_weight/mean":-2.523884177207947e-07,"train/train/tensor_param_model_layers_76_mlp_waleed_W_u_weight/max_abs":0.2470703125,"train/train/layer_model_layers_45/grad/mean":-1.232111797480613e-07,"train/train/tensor_act_model_layers_69_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/std":0.0001640585437990054,"train/train/tensor_param_model_layers_9_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_mlp_waleed_W_u_weight/std":0.02587890625,"train/train/tensor_grad_model_layers_24_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/std":6.124422929097929e-06,"train/train/tensor_act_model_layers_24_self_attn_v_proj/norm":2338.4212184451158,"train/train/tensor_act_model_layers_25_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_38/grad/mean":1.1091492488957047e-07,"train/train/tensor_param_model_layers_51_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn/max_abs":6.75,"train/train/tensor_act_model_layers_58_mlp_waleed_W_u/mean":-0.0085906982421875,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/std":0.02294921875,"train/train/tensor_param_model_layers_71_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/mean":7.418566383421421e-08,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/norm":0.012079835055102243,"train/train/tensor_act_model_layers_23_mlp_down_proj/mean":0.0002722740173339844,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/norm":0.002355333848667708,"train/train/tensor_act_model_layers_30_mlp_waleed_W_g/mean":0.00140380859375,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/max_abs":0.0008697509765625,"train/train/tensor_act_model_layers_35_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_7_input_layernorm/norm":5792.613403324289,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp_waleed_W_g/max_abs":2.78125,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/max_abs":0.001007080078125,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_64_mlp_waleed_W_g_weight/max_abs":0.1796875,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/norm":0.059329437040419326,"train/train/tensor_act_model_layers_87_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_waleed_W_g_weight/norm":0.0472694332385641,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/std":3.6905100604931734e-05,"train/train/tensor_act_model_layers_91_mlp_waleed_W_g/std":0.8603537578898445,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/max_abs":0.0380859375,"train/train/tensor_act_model_layers_29_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn_k_proj/norm":6686.949829927352,"train/train/tensor_act_model_layers_75_mlp_waleed_W_g/max_abs":3.359375,"train/train/tensor_param_model_layers_20_mlp_waleed_W_u_weight/std":0.0233154296875,"train/train/tensor_act_model_layers_79_self_attn/norm":1744.0613024313504,"train/train/tensor_act_model_layers_75_self_attn_k_proj/norm":6236.4529641462195,"train/train/tensor_grad_model_layers_66_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/mean":-0.0005869865417480469,"train/train/tensor_param_model_layers_28_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_input_layernorm/mean":0.001323699951171875,"train/train/tensor_param_model_layers_29_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_mlp_waleed_W_u_weight/mean":4.6253204345703125e-05,"train/train/tensor_param_model_layers_68_mlp_waleed_W_u_weight/std":0.032470703125,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/std":0.0458984375,"train/train/layer_model_layers_29/act/mean":-0.0034921765327453613,"train/train/tensor_grad_model_layers_67_mlp_waleed_W_g_weight/mean":-8.242204785346985e-08,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/max_abs":0.22265625,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/mean":-0.0003986358642578125,"train/train/tensor_act_model_layers_80_post_attention_layernorm/std":1.0000003281747383,"train/train/tensor_act_model_layers_51_self_attn_k_proj/norm":5845.762080895191,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/max_abs":0.000514984130859375,"train/train/tensor_grad_model_layers_62_mlp_waleed_W_g_weight/norm":0.017061591637714004,"train/train/layer__model_layers_62/param/max_abs":1,"train/train/layer_model_layers_85/grad/max_abs":0.00144195556640625,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/norm":0.014130462556551746,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_29/grad/std":4.227685938346116e-05,"train/train/tensor_grad_model_layers_50_mlp_waleed_W_g_weight/std":4.4008454237591906e-05,"train/train/layer_model_layers_8/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp_waleed_W_u/norm":2178.985358668976,"train/train/tensor_act_model_layers_43_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_input_layernorm/norm":5792.603759766598,"train/train/tensor_act_model_layers_85_mlp_down_proj/norm":2422.5119902145016,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/mean":9.059906005859375e-06,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/norm":2.734375,"train/train/tensor_act_model_layers_47_self_attn_v_proj/max_abs":1.546875,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/max_abs":0.002349853515625,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/norm":0.021638749106729263,"train/train/tensor_param_model_layers_31_mlp_waleed_W_u_weight/norm":4.4375,"train/train/tensor_act_model_layers_17_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/norm":0.018746637996619558,"train/train/layer__model_layers_72/param/mean":0.0016138609411563963,"train/train/tensor_act_model_layers_17/frac_near_user_limit":0,"train/train/layer_model_layers_48/grad/mean":2.6719223318531436e-08,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_58/act/std":0.8652584227965752,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/norm":7.1875,"train/train/tensor_act_model_layers_73_self_attn_v_proj/std":0.5078126294681494,"train/train/layer_model_layers_41/grad/std":3.648327720667362e-05,"train/train/tensor_act_model_layers_7_mlp/norm":852.9502117896599,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_waleed_W_u_weight/norm":4.46875,"train/train/tensor_param_model_layers_8_mlp_waleed_W_g_weight/norm":4.15625,"train/train/tensor_act_model_layers_91_self_attn_k_proj/max_abs":5.75,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/mean":-0.00010538101196289062,"train/train/layer_model_layers_35/act/mean":-0.0030471421778202057,"train/train/tensor_act_model_layers_67_self_attn_o_proj/mean":-0.0007352828979492188,"train/train/tensor_grad_model_layers_52_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/mean":-3.3855438232421875e-05,"train/train/tensor_act_model_layers_12_self_attn_v_proj/norm":2362.4100688209996,"train/train/tensor_act_model_layers_77_post_attention_layernorm/norm":5792.615722659162,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/max_abs":0.00093841552734375,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/std":0.0235595703125,"train/train/tensor_param_model_layers_20_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_waleed/std":0.21875008945684113,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/mean":-1.609325408935547e-05,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/max_abs":0.2021484375,"train/train/tensor_act_model_layers_90_self_attn_k_proj/max_abs":8.1875,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/norm":0.003157546389138767,"train/train/tensor_grad_model_layers_7_mlp_waleed_W_u_weight/mean":-2.1141022443771362e-07,"train/train/tensor_act_model_layers_85_mlp_waleed/norm":3339.707876765652,"train/train/layer_model_layers_11/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_11/param/max_abs":1,"train/train/layer_model_layers_12/act/max_abs":26.25,"train/train/tensor_act_model_layers_77_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_o_proj/norm":491.34969060766144,"train/train/tensor_grad_model_layers_77_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/max_abs":0.13671875,"train/train/tensor_param_model_layers_41_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_91_mlp_waleed_W_g_weight/max_abs":0.263671875,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/std":0.03173828125,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/mean":8.393544703722e-08,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_mlp_waleed_W_u_weight/max_abs":0.255859375,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/norm":0.02624696237154631,"train/train/tensor_act_model_layers_56_mlp_down_proj/frac_near_user_limit":0,"train/train/layer__model_layers_57/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_18_self_attn_q_proj/mean":0.01922607421875,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/std":0.0244140625,"train/train/tensor_act_model_layers_33_mlp/max_abs":0.337890625,"train/train/tensor_param_model_layers_72_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_85_mlp_down_proj/max_abs":3.1875,"train/train/tensor_act_model_layers_77_input_layernorm/std":1.0000005733452957,"train/train/tensor_act_model_layers_66_mlp_waleed_W_g/std":0.41015641824117116,"train/train/layer_model_layers_53/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/norm":2.984375,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/max_abs":0.000583648681640625,"train/train/tensor_act_model_layers_16_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_23/act/mean":0.007417440414428711,"train/train/tensor_act_model_layers_86_self_attn/mean":-0.00782012939453125,"train/train/tensor_act_model_layers_56_mlp_waleed/max_abs":3.640625,"train/train/tensor_act_model_layers_10_mlp_waleed_W_g/norm":2209.523806590968,"train/train/tensor_param_model_layers_70_mlp_waleed_W_u_weight/mean":6.198883056640625e-05,"train/train/global/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/norm":0.019901579448186286,"train/train/tensor_act_model_layers_49_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_44/act/max_abs":24.625,"train/train/tensor_act_model_layers_30_input_layernorm/mean":0.00025081634521484375,"train/train/tensor_act_model_layers_78_mlp_down_proj/norm":1168.3149186528387,"train/train/tensor_act_model_layers_2_mlp_waleed/std":0.43701479180816555,"train/train/tensor_act_model_layers_63/std":2.5976904994097825,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/norm":0.011865138614723368,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/max_abs":0.000591278076171875,"train/train/tensor_grad_model_layers_87_mlp_waleed_W_u_weight/norm":0.033209856136415895,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/norm":0.004169594988640427,"train/train/tensor_act_model_layers_1_mlp_down_proj/max_abs":25.375,"train/train/tensor_grad_model_layers_57_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp/max_abs":0.64453125,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/std":3.9945826474456016e-05,"train/train/tensor_param_model_layers_74_mlp_waleed_W_u_weight/max_abs":0.1708984375,"train/train/tensor_act_model_layers_36_self_attn_k_proj/max_abs":6,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_lm_head/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp_waleed/std":0.13061592109651585,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/mean":1.537799835205078e-05,"train/train/tensor_grad_model_layers_56_mlp_waleed_W_u_weight/mean":-2.644956111907959e-07,"train/train/tensor_act_model_layers_82_mlp_down_proj/norm":2058.390705624139,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/std":6.403260262105901e-05,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/mean":1.6927719116210938e-05,"train/train/tensor_param_model_layers_0_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_13_mlp_waleed_W_u_weight/max_abs":0.11181640625,"train/train/tensor_param_model_layers_10_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_42_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_92_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_k_proj/norm":4652.744587479875,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/std":0.0250244140625,"train/train/layer_model_layers_7/act/max_abs":26.875,"train/train/tensor_param_model_layers_68_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_79_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/std":0.021728515625,"train/train/layer__model_layers_49/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/norm":0.007974611632239111,"train/train/tensor_act_model_layers_24_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_22/param/max_abs":1,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/max_abs":0.185546875,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/norm":2.75,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/mean":4.9709342420101166e-08,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/max_abs":0.00049591064453125,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/max_abs":0.0001964569091796875,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/mean":-3.9696693420410156e-05,"train/train/tensor_act_model_layers_92_post_attention_layernorm/max_abs":5.90625,"train/train/layer_model_layers_64/grad/mean":1.074464384367611e-07,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/mean":-1.9818544387817383e-06,"train/train/tensor_param_model_layers_69_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_input_layernorm/norm":5791.63415528115,"train/train/tensor_act_model_layers_18_mlp_down_proj/max_abs":0.50390625,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/mean":0.0002460479736328125,"train/train/layer_model_layers_90/grad/mean":-2.911944936106618e-07,"train/train/tensor_param_model_layers_67_mlp_waleed_W_u_weight/norm":5.96875,"train/train/tensor_param_model_layers_51_mlp_waleed_W_g_weight/norm":5.0625,"train/train/tensor_act_model_layers_26_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn/std":0.22876453268411048,"train/train/layer_model_layers_86/act/norm":25721.474844320444,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_55/param/norm":20.349035651674626,"train/train/tensor_grad_model_layers_82_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/norm":5.28125,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_88/grad/norm":0.0716602000438861,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/mean":-4.7660432755947113e-07,"train/train/tensor_act_model_layers_82_mlp/norm":2058.390705624139,"train/train/tensor_act_model_layers_56_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_mlp/mean":0.00022912025451660156,"train/train/tensor_act_model_layers_81_self_attn_q_proj/norm":7386.760812284053,"train/train/tensor_act_model_layers_91_self_attn/norm":3620.590754566589,"train/train/tensor_act_model_layers_4_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_mlp_waleed_W_g_weight/std":3.4751315851887606e-05,"train/train/tensor_act_model_layers_40_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_mlp_waleed_W_g_weight/mean":0.0003910064697265625,"train/train/tensor_act_model_layers_67_mlp_waleed_W_g/max_abs":3.125,"train/train/tensor_act_model_layers_43_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28/mean":0.0003566741943359375,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_waleed_W_u_weight/norm":6.6875,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/max_abs":0.244140625,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/mean":7.226481102406979e-08,"train/train/tensor_act_model_layers_6_self_attn_o_proj/mean":0.00013178586959838867,"train/train/tensor_act_model_layers_12_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_v_proj/std":0.5791042053518571,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_51/grad/norm":0.04317725508457254,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/std":0.023193359375,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/max_abs":8.726119995117188e-05,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_35_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/max_abs":0.000934600830078125,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_25_mlp_waleed_W_u_weight/max_abs":0.13671875,"train/train/tensor_param_model_layers_32_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/max_abs":0.2138671875,"train/train/tensor_act_model_layers_45_input_layernorm/mean":0.004665374755859375,"train/train/tensor_act_model_layers_3_mlp_waleed_W_g/norm":3837.687998166586,"train/train/tensor_param_model_layers_72_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/mean":4.190951585769653e-09,"train/train/tensor_act_model_layers_76/std":2.875022471632612,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/norm":0.03882641978520286,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/norm":5,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn_o_proj/norm":1065.7751087551435,"train/train/tensor_param_model_layers_6_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_90/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_mlp_waleed_W_g_weight/norm":0.014285551808914885,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/max_abs":0.09423828125,"train/train/tensor_param_model_layers_32_mlp_waleed_W_u_weight/mean":2.7894973754882812e-05,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/max_abs":0.00049591064453125,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/std":0.029296875,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/norm":0.0011846032421098216,"train/train/tensor_act_model_layers_0_input_layernorm/mean":-0.00522613525390625,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/mean":0.00020599365234375,"train/train/tensor_act_model_layers_62_self_attn_v_proj/std":0.3750001176570668,"train/train/tensor_act_model_layers_81_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn_q_proj/mean":0.00836181640625,"train/train/tensor_act_model_layers_92_mlp_waleed/max_abs":16.875,"train/train/tensor_act_model_layers_53_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/norm":7.09375,"train/train/tensor_act_model_layers_53_self_attn_v_proj/max_abs":3.65625,"train/train/tensor_param_model_layers_72_mlp_waleed_W_u_weight/norm":6.21875,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/max_abs":0.271484375,"train/train/tensor_act_model_layers_69_self_attn_o_proj/max_abs":2.875,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/max_abs":0.000640869140625,"train/train/layer__model_layers_57/param/norm":21.027128589485297,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/max_abs":0.000598907470703125,"train/train/tensor_grad_model_layers_53_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_mlp_waleed_W_g_weight/norm":4.65625,"train/train/tensor_act_model_layers_49_input_layernorm/mean":-0.0003376007080078125,"train/train/layer_model_layers_76/grad/mean":-7.037696677046521e-08,"train/train/tensor_act_model_layers_52_mlp_down_proj/std":0.08142193545229959,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/norm":0.011645965247297115,"train/train/tensor_grad_model_layers_13_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_q_proj/max_abs":5.125,"train/train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/std":8.213963049671325e-05,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_mlp_waleed_W_g_weight/mean":-1.0710209608078003e-07,"train/train/tensor_grad_model_layers_10_mlp_waleed_W_u_weight/max_abs":0.0029144287109375,"train/train/layer_model_layers_69/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer_model_layers_54/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_k_proj/max_abs":4.28125,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp/mean":0.0007200241088867188,"train/train/tensor_act_model_layers_71_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/max_abs":0.00028228759765625,"train/train/tensor_param_model_layers_42_mlp_waleed_W_g_weight/norm":4.6875,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/std":0.04296875,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_input_layernorm/std":1.0000000135041773,"train/train/tensor_act_model_layers_66_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn_k_proj/std":0.888674314726111,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_q_proj/norm":6164.343527980554,"train/train/tensor_param_model_layers_23_mlp_waleed_W_u_weight/std":0.023193359375,"train/train/tensor_act_model_layers_30_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_waleed_W_u_weight/std":0.0294189453125,"train/train/tensor_act_model_layers_69/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/std":0.0250244140625,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/std":3.761185002337471e-05,"train/train/tensor_act_model_layers_89_mlp/mean":-0.00921630859375,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/std":0.0245361328125,"train/train/tensor_grad_model_layers_55_mlp_waleed_W_u_weight/mean":-1.835869625210762e-07,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_o_proj/max_abs":1.203125,"train/train/tensor_act_model_layers_33_self_attn_q_proj/norm":5167.834198832188,"train/train/tensor_act_model_layers_62_self_attn_q_proj/max_abs":6.3125,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/std":6.737637614514346e-05,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/mean":2.0532752387225628e-08,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/norm":0.015458079665237115,"train/train/tensor_grad_model_layers_1_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/norm":0.03644836380298984,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/std":1.3365870628545625e-05,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/mean":0.000514984130859375,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/norm":0.0008454396774132009,"train/train/tensor_act_model_layers_16/max_abs":26.25,"train/train/tensor_act_model_layers_69_input_layernorm/norm":5792.609741211445,"train/train/tensor_act_model_layers_66_input_layernorm/max_abs":5.125,"train/train/tensor_param_model_layers_40_mlp_waleed_W_g_weight/std":0.0263671875,"train/train/tensor_grad_model_layers_11_mlp_waleed_W_g_weight/mean":-1.2350938050076365e-08,"train/train/tensor_act_model_layers_81_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/norm":0.008351228791181255,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/std":3.838650858296913e-05,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn/max_abs":2.78125,"train/train/tensor_act_model_layers_82_self_attn_v_proj/norm":3396.2889420215743,"train/train/layer_model_layers_75/grad/std":7.179900118509137e-05,"train/train/tensor_act_model_layers_93_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/mean":2.1457672119140625e-06,"train/train/tensor_act_model_layers_9_mlp_waleed_W_u/std":0.2773437827405776,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_mlp_waleed/mean":-0.00122833251953125,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_v_proj/max_abs":3.453125,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/norm":4.96875,"train/train/tensor_act_model_layers_81_mlp_down_proj/max_abs":3.28125,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/mean":-3.1250237952917814e-08,"train/train/layer_model_layers_55/grad/mean":-5.066409335121536e-08,"train/train/tensor_grad_model_layers_23_mlp_waleed_W_u_weight/norm":0.012361163761024408,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/max_abs":0.123046875,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/norm":5.6875,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/max_abs":0.0004138946533203125,"train/train/tensor_act_model_layers_52_self_attn_v_proj/std":0.3339844608341631,"train/train/tensor_act_model_layers_27_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn_v_proj/norm":2502.1688917060974,"train/train/tensor_act_model_layers_64_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp_waleed/norm":2614.9607959298296,"train/train/tensor_act_model_layers_76_mlp_down_proj/std":0.18750005419132446,"train/train/tensor_act_model_layers_66_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_35/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/mean":8.172355592250824e-08,"train/train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_q_proj/max_abs":7.21875,"train/train/tensor_param_model_layers_63_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_mlp_waleed_W_g_weight/mean":-0.0001964569091796875,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/norm":0.013842572945786159,"train/train/tensor_act_model_layers_34/norm":15466.427364660793,"train/train/tensor_act_model_layers_4_mlp_waleed_W_u/mean":0.0018901824951171875,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/norm":0.0012439890968354362,"train/train/tensor_param_model_layers_25_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/max_abs":0.00016021728515625,"train/train/tensor_act_model_layers_45_self_attn_o_proj/norm":1128.4717260247437,"train/train/layer__model_layers_34/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_waleed_W_u_weight/std":0.0247802734375,"train/train/tensor_act_model_layers_33/std":2.6718986581920765,"train/train/tensor_grad_model_layers_48_mlp_waleed_W_g_weight/std":4.2320241840509276e-05,"train/train/tensor_act_model_layers_67_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/max_abs":0.001312255859375,"train/train/tensor_param_model_layers_20_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_52_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/std":0.028076171875,"train/train/tensor_grad_model_layers_80_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer_model_layers_90/act/std":1.3113480822178438,"train/train/tensor_act_model_layers_90/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/max_abs":0.29296875,"train/train/tensor_param_model_layers_40_mlp_waleed_W_u_weight/mean":-2.6702880859375e-05,"train/train/tensor_act_model_layers_39/mean":0.005527496337890625,"train/train/tensor_act_model_layers_88_mlp/std":0.5517605304651074,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/mean":-4.353933036327362e-07,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/std":2.9028540645357023e-05,"train/train/layer__model_layers_19/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/mean":-1.7159618437290192e-07,"train/train/tensor_act_model_layers_88_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_74/grad/norm":0.07555379523750692,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/max_abs":0.2265625,"train/train/layer_model_layers_0/act/max_abs":20.75,"train/train/tensor_act_model_layers_73_mlp_waleed/std":0.20678755085001754,"train/train/tensor_act_model_layers_12_self_attn_v_proj/std":0.4082031803267957,"train/train/layer__model_layers_48/param/mean":0.0015841080878342556,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/max_abs":0.181640625,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/mean":4.445610102266073e-09,"train/train/layer_model_layers_29/act/std":0.856836382407818,"train/train/tensor_act_model_layers_26_mlp_down_proj/max_abs":0.53515625,"train/train/tensor_param_model_layers_62_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_70/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_waleed_W_u_weight/norm":6.03125,"train/train/layer_model_layers_20/grad/std":4.448039717776601e-05,"train/train/tensor_act_model_layers_55_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/layer_model_layers_13/act/std":0.9437821958829371,"train/train/tensor_act_model_layers_5_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/max_abs":0.150390625,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/norm":0.002488609793814833,"train/train/layer__model_layers_43/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_k_proj/norm":5413.094727998701,"train/train/layer__model_layers_69/param/norm":21.5335913549186,"train/train/tensor_act_model_layers_26/mean":0.00704193115234375,"train/train/layer_model_layers_84/act/max_abs":29,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_21/grad/max_abs":0.002227783203125,"train/train/layer_model_layers_6/grad/std":0.00013232797252078996,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/std":0.021484375,"train/train/tensor_act_model_layers_51_mlp_waleed/mean":0.002960205078125,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/norm":5.21875,"train/train/tensor_grad_model_layers_74_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn/norm":3511.495258268671,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_k_proj/std":0.9921875544420362,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/max_abs":0.00140380859375,"train/train/tensor_act_model_layers_30_mlp_waleed_W_g/std":0.2778333294549858,"train/train/tensor_grad_model_layers_46_mlp_waleed_W_u_weight/mean":-1.05355866253376e-08,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/max_abs":0.19921875,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/mean":-0.00041961669921875,"train/train/tensor_act_model_layers_72_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn_o_proj/std":0.3515652745852656,"train/train/layer_model_layers_60/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_waleed_W_u_weight/norm":0.27592168155661007,"train/train/tensor_grad_model_layers_38_mlp_waleed_W_u_weight/std":3.58257494437077e-05,"train/train/tensor_act_model_layers_29_post_attention_layernorm/norm":5792.612426761531,"train/train/tensor_act_model_layers_37_self_attn_o_proj/mean":-0.0005855560302734375,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/norm":0.019948531063335793,"train/train/tensor_act_model_layers_72_self_attn/mean":-0.0059051513671875,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/norm":4.3125,"train/train/tensor_grad_model_layers_72_mlp_waleed_W_u_weight/max_abs":0.000553131103515625,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/std":0.03955078125,"train/train/tensor_act_model_layers_35_self_attn_v_proj/mean":0.0012674331665039062,"train/train/tensor_act_model_layers_46_self_attn_o_proj/mean":0.0003108978271484375,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/std":3.355971587371967e-05,"train/train/tensor_act_model_layers_7_mlp_down_proj/std":0.1472190254589959,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/std":0.00010789785827667963,"train/train/tensor_act_model_layers_24_self_attn/mean":0.002655029296875,"train/train/tensor_param_model_layers_35_mlp_waleed_W_g_weight/mean":-0.00012683868408203125,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/norm":0.010500796645460926,"train/train/tensor_act_model_layers_77_self_attn_q_proj/max_abs":6.9375,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/std":0.0361328125,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/mean":2.1047890186309814e-07,"train/train/tensor_grad_model_layers_6_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_mlp_waleed_W_u_weight/norm":0.023592673649772865,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/max_abs":0.000743865966796875,"train/train/tensor_act_model_layers_23_self_attn_o_proj/std":0.3632838940652383,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/std":9.081985268679273e-05,"train/train/tensor_param_model_layers_50_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/norm":852.9502117896599,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/norm":0.0012539682228157286,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/max_abs":0.00022792816162109375,"train/train/tensor_act_model_layers_60_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/std":4.7463595244269676e-05,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/mean":-8.102506399154663e-07,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/norm":5.25,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/std":0.0001950526691532315,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/std":0.033935546875,"train/train/tensor_grad_model_layers_42_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/norm":0.0007952477349022814,"train/train/tensor_param_model_layers_42_mlp_waleed_W_g_weight/std":0.02587890625,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp_waleed_W_u/max_abs":3.453125,"train/train/tensor_grad_model_layers_51_mlp_waleed_W_u_weight/std":3.9823559226298144e-05,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/norm":0.00967883645566045,"train/train/tensor_act_model_layers_42_self_attn_o_proj/norm":712.9822687184986,"train/train/layer__model_layers_81/param/max_abs":1,"train/train/tensor_grad_model_layers_12_mlp_waleed_W_g_weight/std":5.947444859117949e-05,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/mean":-1.5173573046922684e-06,"train/train/tensor_act_model_layers_83_self_attn_v_proj/norm":3212.533137128756,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/norm":2.953125,"train/train/layer__model_layers_38/param/std":0.05012911827577111,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/mean":-7.776543498039246e-07,"train/train/layer_model_layers_69/act/mean":0.011619821190834045,"train/train/tensor_act_model_layers_92_mlp_waleed_W_g/max_abs":6.1875,"train/train/tensor_grad_model_layers_44_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/norm":0.02169823411687025,"train/train/tensor_act_model_layers_58_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/norm":0.001515184116206112,"train/train/tensor_act_model_layers_65_mlp_waleed_W_u/mean":0.00647735595703125,"train/train/tensor_act_model_layers_58_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_mlp_waleed_W_g_weight/mean":6.535265129059553e-08,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/mean":8.629285730421543e-09,"train/train/tensor_act_model_layers_56_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_input_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_55/param/mean":0.001621232203871895,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23/std":2.8047421790082034,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_v_proj/std":0.6093815870418084,"train/train/tensor_act_model_layers_9_self_attn_v_proj/std":0.4082031377622383,"train/train/tensor_act_model_layers_40_mlp_waleed_W_g/mean":0.00460052490234375,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_v_proj/std":0.539062770282566,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/std":5.987436675213214e-05,"train/train/tensor_grad_model_layers_4_mlp_waleed_W_g_weight/std":0.00013072819980130786,"train/train/tensor_grad_model_layers_56_mlp_waleed_W_u_weight/max_abs":0.00057220458984375,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/mean":5.529727786779404e-09,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/max_abs":0.2138671875,"train/train/tensor_act_model_layers_52_input_layernorm/mean":0.0007729530334472656,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/mean":-2.658367156982422e-05,"train/train/tensor_grad_model_layers_83_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_55/grad/std":4.431712686862616e-05,"train/train/layer_model_layers_27/act/std":0.8819150466021699,"train/train/tensor_act_model_layers_33_mlp_waleed/mean":-0.0005502700805664062,"train/train/tensor_grad_model_layers_72_mlp_waleed_W_g_weight/norm":0.02163806046256431,"train/train/tensor_act_model_layers_70_self_attn_k_proj/norm":6063.624746454484,"train/train/layer_model_layers_81/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_mlp_waleed_W_g_weight/std":5.006823793196686e-05,"train/train/tensor_act_model_layers_12_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49/mean":0.002735137939453125,"train/train/tensor_act_model_layers_34_self_attn_o_proj/max_abs":1.4609375,"train/train/layer__model_layers_69/param/max_abs":1,"train/train/tensor_act_model_layers_54/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn_k_proj/max_abs":6.8125,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/mean":-8.314382284879684e-07,"train/train/tensor_param_model_layers_62_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_2/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_down_proj/mean":0.0078582763671875,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/norm":0.013673577883800412,"train/train/tensor_act_model_layers_45_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/max_abs":0.1533203125,"train/train/tensor_act_model_layers_54/std":2.6015994256292423,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/norm":4.65625,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/mean":-3.80445271730423e-07,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/norm":5.84375,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/norm":0.01383163569223325,"train/train/tensor_act_model_layers_70_mlp_down_proj/norm":741.2930685159407,"train/train/tensor_grad_model_layers_85_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_56/grad/frac_near_user_limit":0,"train/train/layer_model_layers_92/act/mean":-0.012144684791564941,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/mean":1.3009412214159966e-08,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/mean":3.573950380086899e-08,"train/train/layer_model_layers_32/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_52_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_mlp_waleed_W_g_weight/norm":4.28125,"train/train/tensor_act_model_layers_41_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_33_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_waleed_W_g/norm":1998.390998340258,"train/train/tensor_act_model_layers_52_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp_waleed/mean":-0.0033111572265625,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_73/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_8/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/norm":6.125,"train/train/tensor_act_model_layers_49_mlp/norm":564.1529521831867,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/mean":3.0617229640483856e-07,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/norm":0.0007899970027516454,"train/train/tensor_act_model_layers_40_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/mean":-2.9336661100387573e-07,"train/train/tensor_act_model_layers_89_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_k_proj/mean":0.025787353515625,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/max_abs":0.248046875,"train/train/tensor_act_model_layers_75_self_attn_k_proj/mean":0.04132080078125,"train/train/tensor_grad_model_layers_59_mlp_waleed_W_g_weight/norm":0.01638893394365242,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/norm":7.5,"train/train/tensor_act_model_layers_87_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp_waleed_W_u/std":0.3105469090840614,"train/train/layer__model_layers_24/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_waleed_W_g_weight/norm":5.34375,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/norm":0.008320700363288379,"train/train/tensor_act_model_layers_60_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_k_proj/mean":0.044189453125,"train/train/tensor_act_model_layers_42/norm":15177.550499130766,"train/train/layer_model_layers_31/grad/std":4.668779719077914e-05,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/std":0.04443359375,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/norm":3.015625,"train/train/tensor_act_model_layers_70_self_attn_q_proj/max_abs":8.25,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/max_abs":0.000537872314453125,"train/train/tensor_act_model_layers_88_mlp_waleed_W_g/max_abs":4.90625,"train/train/tensor_act_model_layers_67_self_attn_v_proj/mean":0.007415771484375,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/norm":0.013596086246510548,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_o_proj/std":0.2905318737365574,"train/train/tensor_param_model_layers_86_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn_q_proj/std":1.2343753923343084,"train/train/tensor_act_model_layers_71_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/std":9.81738132596482e-05,"train/train/tensor_act_model_layers_9_self_attn_o_proj/std":0.09594770856383782,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/mean":-9.268522262573242e-06,"train/train/tensor_act_model_layers_32_input_layernorm/norm":5792.617675782248,"train/train/layer_model_layers_85/act/std":1.12269814868188,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/std":0.0005808463867327013,"train/train/tensor_act_model_layers_15_mlp_waleed_W_u/max_abs":2.09375,"train/train/tensor_act_model_layers_18_mlp/norm":387.1145685451204,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/max_abs":0.000579833984375,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/max_abs":0.091796875,"train/train/tensor_act_model_layers_15_input_layernorm/mean":-0.005462646484375,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/max_abs":0.0003757476806640625,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/std":0.00014600119290431146,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/mean":-1.7240643501281738e-05,"train/train/tensor_act_model_layers_21_mlp_waleed_W_g/mean":0.001346588134765625,"train/train/tensor_act_model_layers_90_self_attn_o_proj/std":0.6064731943551857,"train/train/tensor_act_model_layers_86_mlp_waleed_W_u/std":0.672854841526053,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/max_abs":0.220703125,"train/train/layer__model_layers_64/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_input_layernorm/mean":-0.006591796875,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_post_attention_layernorm/std":1.0000000121071937,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/norm":5.25,"train/train/layer__model_layers_67/param/mean":0.0014985184215718238,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/norm":0.00829462215581804,"train/train/tensor_act_model_layers_41_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp_waleed_W_u/std":0.29101563780099726,"train/train/tensor_act_model_layers_56_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/max_abs":0.0002651214599609375,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/std":0.0263671875,"train/train/tensor_param_model_layers_88_mlp_waleed_W_g_weight/max_abs":0.26953125,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/std":5.4126040590561785e-05,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/norm":6.21875,"train/train/tensor_param_model_layers_68_mlp_waleed_W_u_weight/mean":-6.580352783203125e-05,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/mean":2.7894973754882812e-05,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/max_abs":0.001251220703125,"train/train/tensor_act_model_layers_41_mlp_waleed_W_u/mean":0.005126953125,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/max_abs":0.1884765625,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/std":0.0380859375,"train/train/tensor_act_model_layers_71_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/layer_model_layers_57/grad/norm":0.03897607488811292,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_33/grad/std":3.0427492162789915e-05,"train/train/tensor_act_model_layers_70_input_layernorm/mean":0.003917694091796875,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/std":0.023681640625,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/std":3.397969262827854e-05,"train/train/tensor_act_model_layers_8_mlp_waleed_W_g/max_abs":2.25,"train/train/tensor_act_model_layers_74/norm":16357.868415790137,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/max_abs":0.00022125244140625,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/max_abs":0.12158203125,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_mlp_waleed_W_g_weight/mean":-1.9185245037078857e-07,"train/train/tensor_act_model_layers_11_mlp_waleed/max_abs":2.734375,"train/train/tensor_act_model_layers_45_self_attn/max_abs":2.125,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/mean":-6.341934204101562e-05,"train/train/tensor_grad_model_layers_54_mlp_waleed_W_g_weight/max_abs":0.00064849853515625,"train/train/tensor_act_model_layers_15_mlp_waleed_W_u/std":0.24316408003990786,"train/train/tensor_act_model_layers_5_self_attn_v_proj/norm":2301.840316065775,"train/train/layer_model_layers_15/act/max_abs":26.25,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/mean":2.8638169169425964e-07,"train/train/layer_model_layers_84/act/std":1.2164617287684611,"train/train/layer_model_layers_16/act/max_abs":26.25,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/max_abs":0.000888824462890625,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/max_abs":0.349609375,"train/train/tensor_act_model_layers_16_input_layernorm/norm":5792.611206061956,"train/train/tensor_act_model_layers_36_self_attn/std":0.23293333455581494,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/norm":0.001087302486423617,"train/train/tensor_param_model_layers_81_mlp_waleed_W_u_weight/norm":7.28125,"train/train/layer_model_layers_46/grad/frac_near_dtype_limit":0,"train/train/layer__model_layers_60/param/norm":20.44631403976032,"train/train/tensor_act_model_layers_17_mlp/max_abs":0.46484375,"train/train/tensor_act_model_layers_88_self_attn/max_abs":3.984375,"train/train/tensor_param_model_layers_83_mlp_waleed_W_g_weight/norm":7.40625,"train/train/tensor_act_model_layers_86_mlp_down_proj/std":0.44775907765431594,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_21/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn_q_proj/std":1.0488336623345793,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/mean":-7.689232006669044e-08,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/mean":-2.6402994990348816e-07,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64/std":2.6250253436927875,"train/train/tensor_act_model_layers_73_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/norm":4.84375,"train/train/tensor_act_model_layers_28_input_layernorm/norm":5792.611206061462,"train/train/layer_model_layers_88/grad/max_abs":0.0013427734375,"train/train/tensor_act_model_layers_19_post_attention_layernorm/max_abs":5.21875,"train/train/tensor_act_model_layers_76_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_waleed_W_u_weight/mean":-4.01865690946579e-07,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_47_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_mlp_waleed_W_u_weight/mean":-4.6711647883057594e-08,"train/train/tensor_act_model_layers_64_self_attn_q_proj/std":1.044927379558004,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/max_abs":0.255859375,"train/train/tensor_act_model_layers_52_self_attn_q_proj/max_abs":5.03125,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/max_abs":0.00010251998901367188,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/max_abs":0.1201171875,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/max_abs":0.1171875,"train/train/tensor_act_model_layers_64/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn_v_proj/norm":3365.1331438815514,"train/train/layer_model_layers_86/grad/mean":-5.018898807524147e-08,"train/train/tensor_act_model_layers_20_mlp_down_proj/mean":0.0015163421630859375,"train/train/layer__model_layers_59/param/mean":0.0014046551470823481,"train/train/tensor_param_model_layers_11_mlp_waleed_W_g_weight/norm":4.15625,"train/train/tensor_act_model_layers_54_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_mlp_waleed_W_u_weight/max_abs":0.001190185546875,"train/train/tensor_act_model_layers_3_mlp_waleed/frac_near_user_limit":0,"train/train/layer_model_layers_30/grad/norm":0.04345974355340992,"train/train/tensor_act_model_layers_88_self_attn_q_proj/norm":6127.645682807433,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp_waleed/std":0.25585946869387766,"train/train/tensor_act_model_layers_0_post_attention_layernorm/std":1.0000000260770316,"train/train/tensor_act_model_layers_81_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/norm":4.28125,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/mean":-0.00010967254638671875,"train/train/tensor_param_model_layers_65_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_91_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer__model_layers_25/param/std":0.048295952975770724,"train/train/tensor_act_model_layers_87_input_layernorm/std":1.000000120465152,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/std":0.0026248447308511434,"train/train/tensor_act_model_layers_82_mlp_waleed/mean":-0.002471923828125,"train/train/tensor_param_model_layers_54_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/std":0.037841796875,"train/train/layer_model_layers_18/act/max_abs":26.125,"train/train/tensor_grad_model_layers_73_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/max_abs":0.228515625,"train/train/tensor_act_model_layers_19_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_v_proj/norm":4264.167874553034,"train/train/tensor_act_model_layers_44_self_attn_v_proj/mean":-0.0033416748046875,"train/train/tensor_param_model_layers_64_mlp_waleed_W_u_weight/std":0.03125,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/max_abs":0.000469207763671875,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/norm":0.01169050309159204,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/std":0.03955078125,"train/train/tensor_act_model_layers_1_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_down_proj/norm":544.5867110453496,"train/train/tensor_param_model_layers_53_mlp_waleed_W_g_weight/max_abs":0.1533203125,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/std":0.037353515625,"train/train/tensor_grad_model_layers_88_mlp_waleed_W_g_weight/max_abs":0.00095367431640625,"train/train/tensor_act_model_layers_4_self_attn_k_proj/mean":0.021514892578125,"train/train/tensor_act_model_layers_63_self_attn_o_proj/norm":994.2581249560949,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/mean":3.757886588573456e-07,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/std":0.00013838656242790086,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_input_layernorm/norm":5792.61230469248,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/norm":3.6875,"train/train/tensor_act_model_layers_32/std":2.679745187771292,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/norm":1289.0682275198708,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/std":0.00022783312352421344,"train/train/tensor_act_model_layers_61_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_3/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_v_proj/norm":3438.1943151539517,"train/train/tensor_act_model_layers_91_mlp_waleed_W_u/mean":0.03369140625,"train/train/tensor_grad_model_layers_63_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_mlp_waleed_W_g_weight/mean":-0.00010585784912109375,"train/train/tensor_act_model_layers_69_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/norm":0.0060968935328797776,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_input_layernorm/std":1.0000010345460502,"train/train/tensor_act_model_layers_36/std":2.656274174843637,"train/train/tensor_act_model_layers_85_mlp/max_abs":3.1875,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/max_abs":0.0006561279296875,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/std":1.288071668909903e-05,"train/train/tensor_act_model_layers_48_mlp_waleed_W_u/norm":2881.4005005800072,"train/train/tensor_act_model_layers_72_self_attn_v_proj/mean":-0.020294189453125,"train/train/layer__model_layers_32/param/norm":19.2618419339247,"train/train/layer__model_layers_32/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn/std":0.085449952296425,"train/train/layer_model_layers_29/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_v_proj/mean":0.002285003662109375,"train/train/layer_model_layers_21/act/norm":21276.638326695658,"train/train/tensor_act_model_layers_16_self_attn/mean":-0.00014663022011518478,"train/train/tensor_act_model_layers_13_self_attn_q_proj/std":1.2265625223991976,"train/train/tensor_act_model_layers_81_mlp_waleed_W_g/norm":4719.667971718773,"train/train/tensor_act_model_layers_46_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_23/act/max_abs":27,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/std":0.0341796875,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_13/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/norm":2.8125,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/norm":5.15625,"train/train/layer_model_layers_17/grad/std":7.053557661273259e-05,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/mean":-7.18235969543457e-06,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_87_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp_waleed_W_u/mean":-0.01458740234375,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/norm":0.007373737782828568,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/max_abs":0.0001316070556640625,"train/train/tensor_grad_model_layers_81_mlp_waleed_W_u_weight/norm":0.02939227783082171,"train/train/tensor_param_model_layers_64_mlp_waleed_W_g_weight/std":0.03125,"train/train/tensor_act_model_layers_28_post_attention_layernorm/max_abs":5.5,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/max_abs":0.0002231597900390625,"train/train/layer_model_layers_68/grad/std":5.724635284472968e-05,"train/train/tensor_act_model_layers_74_self_attn/norm":2277.4074541225787,"train/train/tensor_act_model_layers_63_post_attention_layernorm/norm":5792.611938483595,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/std":2.4472991586487943e-05,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/std":0.003283891618573935,"train/train/tensor_grad_model_layers_9_mlp_waleed_W_g_weight/max_abs":0.00176239013671875,"train/train/tensor_param_model_layers_10_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_87/grad/max_abs":0.0013275146484375,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/max_abs":0.267578125,"train/train/tensor_act_model_layers_67_mlp_waleed/norm":1490.1791675774116,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/max_abs":0.109375,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/std":0.00011124396371831432,"train/train/tensor_grad_model_layers_28_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/max_abs":0.000247955322265625,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/mean":-6.3478946685791016e-06,"train/train/tensor_grad_model_layers_80_mlp_waleed_W_g_weight/norm":0.027744960108157923,"train/train/tensor_grad_model_layers_67_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/norm":0.0015976114281357216,"train/train/tensor_grad_model_layers_16_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/max_abs":0.0005035400390625,"train/train/tensor_act_model_layers_70_mlp_waleed/std":0.184570326176028,"train/train/layer__model_layers_4/param/std":0.04758422484431024,"train/train/tensor_act_model_layers_75_mlp_down_proj/norm":1085.293398856482,"train/train/layer__model_layers_66/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/norm":0.16202625763576764,"train/train/tensor_act_model_layers_87/mean":-0.0126495361328125,"train/train/tensor_grad_model_layers_83_mlp_waleed_W_g_weight/std":8.357560452285995e-05,"train/train/tensor_param_model_layers_48_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/mean":-8.106231689453125e-05,"train/train/tensor_act_model_layers_75_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_v_proj/mean":-0.011749267578125,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/std":2.107503454578331e-05,"train/train/tensor_act_model_layers_90/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_18_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp/max_abs":25.375,"train/train/tensor_act_model_layers_25_mlp_down_proj/norm":405.5003717256925,"train/train/tensor_act_model_layers_76_self_attn_o_proj/std":0.2470715855833901,"train/train/tensor_grad_model_layers_61_mlp_waleed_W_u_weight/std":5.1737328389263186e-05,"train/train/tensor_param_model_layers_78_mlp_waleed_W_u_weight/mean":8.916854858398438e-05,"train/train/layer__model_layers_80/param/max_abs":1,"train/train/tensor_act_model_layers_73_mlp/norm":891.4172345806643,"train/train/layer_model_layers_54/grad/norm":0.033544805515599734,"train/train/layer_model_layers_45/grad/norm":0.04082061695846324,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/max_abs":0.00141143798828125,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/max_abs":0.001068115234375,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/norm":0.0076340470433304145,"train/train/tensor_act_model_layers_59_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_mlp_waleed_W_g_weight/std":6.398563831382998e-05,"train/train/tensor_act_model_layers_33_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_mlp_waleed/max_abs":4.03125,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/norm":0.0012376561738216346,"train/train/tensor_act_model_layers_77_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36/norm":15394.391400403953,"train/train/tensor_act_model_layers_70_mlp_waleed_W_u/mean":-0.00165557861328125,"train/train/tensor_act_model_layers_32_mlp_waleed_W_u/mean":-0.0023136138916015625,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/mean":-5.9604644775390625e-05,"train/train/tensor_act_model_layers_65_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_o_proj/mean":-0.0006933212280273438,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_mlp_waleed_W_u_weight/norm":4.34375,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_self_attn_k_proj/norm":6025.049431877204,"train/train/tensor_grad_model_layers_14_mlp_waleed_W_u_weight/mean":-4.135654307901859e-08,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/mean":-0.00024318695068359375,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/norm":0.0015456539492078748,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/mean":-0.0012569427490234375,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_56/grad/mean":-8.541579226063863e-08,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/mean":0.00010633468627929688,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/mean":0.00013446807861328125,"train/train/tensor_act_model_layers_89_self_attn_o_proj/norm":2306.719301150095,"train/train/tensor_param_model_layers_0_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn/mean":-0.002460479736328125,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/std":5.189018357127488e-05,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/mean":1.9278377294540405e-06,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp_waleed_W_g/norm":2802.630007572205,"train/train/tensor_grad_model_layers_3_mlp_waleed_W_u_weight/norm":0.0571471617133207,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/mean":-9.190989658236504e-08,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/std":0.0264892578125,"train/train/tensor_act_model_layers_41/norm":15225.529759103847,"train/train/tensor_act_model_layers_36_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_mlp_waleed_W_g_weight/max_abs":0.1474609375,"train/train/layer_model_layers_67/act/norm":21225.1616937293,"train/train/tensor_grad_model_layers_0_mlp_waleed_W_u_weight/max_abs":0.024658203125,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_26_mlp/norm":460.84734975882213,"train/train/tensor_act_model_layers_82_mlp_down_proj/std":0.355468826962033,"train/train/tensor_act_model_layers_62_mlp_waleed_W_u/norm":3126.947418692167,"train/train/tensor_param_model_layers_19_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp_waleed/max_abs":21.25,"train/train/tensor_act_model_layers_37/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/mean":-4.913657903671265e-06,"train/train/tensor_grad_model_layers_19_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/mean":0.00019931793212890625,"train/train/tensor_act_model_layers_22_input_layernorm/mean":-0.004261016845703125,"train/train/tensor_act_model_layers_62_self_attn_k_proj/norm":5341.02188950891,"train/train/layer_model_layers_78/act/max_abs":26,"train/train/tensor_act_model_layers_4_self_attn_q_proj/std":1.4003947828542362,"train/train/tensor_act_model_layers_13_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_input_layernorm_weight/std":0.0004863739013671875,"train/train/tensor_param_model_layers_82_input_layernorm_weight/max_abs":1,"train/train/layer__model_layers_87/param/norm":25.501914756543282,"train/train/tensor_grad_model_layers_24_mlp_waleed_W_u_weight/std":4.125229260901374e-05,"train/train/tensor_act_model_layers_52/max_abs":23.375,"train/train/tensor_act_model_layers_90_mlp_waleed/max_abs":12.6875,"train/train/tensor_act_model_layers_81_mlp_down_proj/norm":1777.9786401867384,"train/train/layer_model_layers_70/grad/std":7.003989550025243e-05,"train/train/tensor_grad_model_layers_73_mlp_waleed_W_u_weight/norm":0.02228075818218569,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_input_layernorm/norm":5792.614379883002,"train/train/tensor_act_model_layers_17_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/std":5.301720708995085e-05,"train/train/tensor_act_model_layers_91_post_attention_layernorm/std":1.0000001072403335,"train/train/tensor_param_model_layers_56_mlp_waleed_W_u_weight/norm":5.1875,"train/train/layer__model_layers_66/param/mean":0.0015327822585559673,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/norm":4.78125,"train/train/tensor_act_model_layers_1_self_attn_o_proj/mean":0.00330352783203125,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/mean":-9.479117579758167e-08,"train/train/tensor_act_model_layers_78_self_attn_q_proj/std":1.107427137132101,"train/train/tensor_act_model_layers_60_mlp/std":0.08667095571559506,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/std":2.97491519631942e-05,"train/train/tensor_act_model_layers_43_mlp_waleed_W_u/norm":2487.408224910763,"train/train/tensor_grad_model_layers_36_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/max_abs":0.251953125,"train/train/tensor_act_model_layers_80_mlp_waleed_W_g/max_abs":3.828125,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/std":3.5942174155794104e-05,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/std":6.41386760640036e-05,"train/train/tensor_act_model_layers_38_post_attention_layernorm/max_abs":5.3125,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/norm":6.875,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/mean":3.995228325948119e-08,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/mean":-4.190951585769653e-09,"train/train/tensor_act_model_layers_14_self_attn_v_proj/std":0.3496095215141799,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/std":0.0284423828125,"train/train/tensor_act_model_layers_79_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_mlp_waleed_W_g/norm":2558.5201104929265,"train/train/tensor_act_model_layers_16_mlp_down_proj/mean":-0.0002503395080566406,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_28/param/std":0.04921357637790385,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp/max_abs":0.85546875,"train/train/layer__model_layers_93/param/std":0.07194982574154551,"train/train/tensor_act_model_layers_28_self_attn_v_proj/std":0.48242192471075385,"train/train/tensor_act_model_layers_63_self_attn_k_proj/std":1.0039139906885388,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_mlp_waleed_W_g_weight/max_abs":0.208984375,"train/train/tensor_act_model_layers_12_self_attn/max_abs":1.1015625,"train/train/tensor_act_model_layers_68_self_attn_o_proj/mean":-0.00035858154296875,"train/train/tensor_act_model_layers_86_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/mean":-3.169989213347435e-07,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_73/param/mean":0.0015777254625341263,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/mean":3.154127625748515e-08,"train/train/tensor_act_model_layers_16_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67/max_abs":25,"train/train/tensor_act_model_layers_18/norm":16451.607478604845,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/norm":4.3125,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/mean":-0.00013256072998046875,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/max_abs":0.00090789794921875,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/norm":0.0052646581543196085,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_post_attention_layernorm/std":1.0000012518932073,"train/train/tensor_act_model_layers_32_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/mean":-0.00012874603271484375,"train/train/tensor_param_model_layers_8_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29/mean":0.0012531280517578125,"train/train/layer_model_layers_45/act/max_abs":24.625,"train/train/tensor_act_model_layers_32_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/std":3.404961959740815e-05,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/max_abs":0.31640625,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_81_input_layernorm/max_abs":5.34375,"train/train/tensor_act_model_layers_52_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/std":0.0380859375,"train/train/tensor_act_model_layers_54_self_attn_o_proj/std":0.10107496798048737,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/max_abs":0.197265625,"train/train/tensor_act_model_layers_43_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/norm":0.025271797893468336,"train/train/tensor_act_model_layers_22_mlp_waleed/mean":0.0019931793212890625,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/mean":7.150229066610336e-07,"train/train/tensor_act_model_layers_67_self_attn_k_proj/std":1.189457948170672,"train/train/tensor_act_model_layers_73_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/std":4.232370019929816e-05,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/mean":0.000118255615234375,"train/train/tensor_act_model_layers_43/norm":15181.795087178753,"train/train/tensor_act_model_layers_49_self_attn_k_proj/max_abs":5.1875,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35/std":2.656274014109776,"train/train/tensor_act_model_layers_90_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/mean":0.0002574920654296875,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_59/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_waleed/norm":1514.9064893569118,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/std":0.0546875,"train/train/tensor_act_model_layers_20_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_29/grad/mean":2.3498187530291806e-08,"train/train/tensor_act_model_layers_93_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/max_abs":0.0002689361572265625,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_q_proj/norm":5975.839146340004,"train/train/tensor_act_model_layers_56_mlp/std":0.08300810673578045,"train/train/tensor_act_model_layers_6_mlp/max_abs":1.8203125,"train/train/tensor_act_model_layers_88_mlp_waleed/std":0.5117188217758172,"train/train/layer_model_layers_93/grad/max_abs":0.001739501953125,"train/train/tensor_act_model_layers_76/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_mlp_down_proj/mean":-0.00021696090698242188,"train/train/tensor_act_model_layers_29_mlp_down_proj/norm":271.246752586825,"train/train/layer_model_layers_2/act/mean":0.003388643264770508,"train/train/tensor_act_model_layers_91_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_waleed_W_u_weight/max_abs":0.12353515625,"train/train/layer_model_layers_67/act/max_abs":25,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/std":6.882587362736948e-05,"train/train/tensor_act_model_layers_65_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_mlp_waleed_W_g_weight/norm":10.3125,"train/train/tensor_act_model_layers_54_self_attn_o_proj/mean":-0.0017547607421875,"train/train/tensor_param_model_layers_42_mlp_waleed_W_u_weight/std":0.0260009765625,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/max_abs":0.115234375,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/max_abs":0.26171875,"train/train/layer_model_layers_64/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/max_abs":0.2099609375,"train/train/tensor_param_model_layers_41_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_mlp_waleed_W_u_weight/std":4.440898569145921e-05,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/max_abs":0.000186920166015625,"train/train/tensor_act_model_layers_12_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_k_proj/max_abs":5.0625,"train/train/layer_model_layers_34/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_4/grad/max_abs":0.0029754638671875,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/std":0.00012482814626439893,"train/train/tensor_grad_model_layers_93_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/mean":-5.044043064117432e-06,"train/train/tensor_act_model_layers_12_mlp_waleed_W_g/std":0.24877968272833673,"train/train/tensor_act_model_layers_34_mlp_waleed/max_abs":2.90625,"train/train/tensor_param_model_layers_58_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_45/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_mlp_waleed_W_u_weight/norm":0.012955960953778024,"train/train/tensor_param_model_layers_33_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_mlp_waleed_W_g/std":0.26171877210153477,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_44/param/std":0.048744661298748715,"train/train/layer__model_layers_81/param/std":0.05843136119040375,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/norm":0.09147522760880546,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_q_proj/max_abs":4.375,"train/train/tensor_act_model_layers_3_mlp/mean":-0.0103302001953125,"train/train/tensor_act_model_layers_30_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/max_abs":0.00018215179443359375,"train/train/tensor_grad_model_layers_70_mlp_waleed_W_g_weight/std":5.495614500536293e-05,"train/train/tensor_act_model_layers_28_mlp_down_proj/max_abs":0.9296875,"train/train/tensor_param_model_layers_57_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_input_layernorm/std":1.0000001586449672,"train/train/tensor_act_model_layers_25_mlp_waleed_W_u/max_abs":2.015625,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/norm":0.001123938258561963,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/norm":5.96875,"train/train/tensor_param_model_layers_12_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_k_proj/mean":0.022247314453125,"train/train/tensor_act_model_layers_53_mlp_down_proj/std":0.0821535980260567,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_11/param/norm":19.39769965622535,"train/train/tensor_act_model_layers_26_input_layernorm/max_abs":5.5,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/max_abs":0.00015926361083984375,"train/train/tensor_act_model_layers_49_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/max_abs":0.146484375,"train/train/tensor_act_model_layers_12_self_attn_k_proj/max_abs":4.75,"train/train/tensor_param_model_layers_34_mlp_waleed_W_g_weight/std":0.0247802734375,"train/train/tensor_act_model_layers_34_input_layernorm/norm":5792.612670899748,"train/train/tensor_act_model_layers_93_input_layernorm/mean":-0.0093231201171875,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_mlp_waleed_W_g_weight/max_abs":0.1259765625,"train/train/tensor_grad_model_layers_36_mlp_waleed_W_u_weight/std":3.6180577770199945e-05,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_45_mlp_waleed_W_g_weight/mean":1.4295801520347595e-07,"train/train/layer_model_layers_20/grad/norm":0.036038173213098504,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/norm":0.04126807134874186,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/max_abs":0.00021839141845703125,"train/train/tensor_param_model_layers_65_mlp_waleed_W_g_weight/max_abs":0.1787109375,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/max_abs":0.000518798828125,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/norm":0.03125441043248275,"train/train/tensor_param_model_layers_5_mlp_waleed_W_u_weight/std":0.0230712890625,"train/train/tensor_act_model_layers_5_input_layernorm/mean":-0.00792694091796875,"train/train/layer_model_layers_79/grad/max_abs":0.00113677978515625,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_k_proj/max_abs":5.1875,"train/train/layer_model_layers_25/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_waleed_W_g_weight/max_abs":0.1259765625,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/norm":3.78125,"train/train/tensor_act_model_layers_64_mlp_down_proj/mean":0.0005345344543457031,"train/train/layer_model_layers_66/act/max_abs":24.625,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/mean":3.790855407714844e-05,"train/train/tensor_act_model_layers_22_self_attn/max_abs":1.2890625,"train/train/tensor_act_model_layers_90_mlp_waleed_W_u/norm":6611.522235009312,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_input_layernorm/std":1.0000011183067838,"train/train/tensor_act_model_layers_51_mlp_down_proj/std":0.08105542862740756,"train/train/tensor_grad_model_layers_54_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/norm":0.025087008462603762,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/max_abs":0.283203125,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/std":0.0247802734375,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/norm":0.011444417313078836,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/norm":4.875,"train/train/tensor_act_model_layers_12_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_k_proj/std":1.4843751916759769,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/max_abs":0.000949859619140625,"train/train/tensor_act_model_layers_52_self_attn_k_proj/norm":4837.48265459515,"train/train/layer_model_layers_63/grad/max_abs":0.00121307373046875,"train/train/tensor_act_model_layers_58_mlp_down_proj/mean":0.0005483627319335938,"train/train/layer__model_layers_50/param/max_abs":1,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/max_abs":0.00020503997802734375,"train/train/tensor_act_model_layers_87_post_attention_layernorm/max_abs":5.96875,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/std":8.471998584554932e-05,"train/train/tensor_act_model_layers_84_mlp_waleed_W_g/mean":0.027313232421875,"train/train/layer_model_layers_25/grad/max_abs":0.0010986328125,"train/train/tensor_act_model_layers_13_mlp_waleed_W_g/max_abs":2.71875,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_mlp_waleed_W_u_weight/std":4.3516193792613184e-05,"train/train/tensor_act_model_layers_56_self_attn/mean":0.0007982254028320312,"train/train/tensor_act_model_layers_8_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/norm":5.4375,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/norm":0.00558334083491386,"train/train/tensor_act_model_layers_47_self_attn_k_proj/norm":4389.0239456988575,"train/train/tensor_act_model_layers_33_post_attention_layernorm/norm":5792.611938478305,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/norm":0.016267329109329598,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/max_abs":0.177734375,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/norm":2.984375,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_mlp_waleed_W_u_weight/max_abs":0.1357421875,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/mean":-7.498601917177439e-08,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_waleed_W_u/norm":2546.3149892115425,"train/train/tensor_act_model_layers_20_mlp_waleed_W_u/norm":1991.721552087432,"train/train/tensor_param_model_layers_26_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_53_self_attn_k_proj/norm":4507.819646160465,"train/train/tensor_act_model_layers_27_mlp_waleed_W_u/mean":-0.003559112548828125,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/mean":1.1723022907972336e-07,"train/train/layer__model_layers_58/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/std":0.00010826779097643019,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_58/grad/std":5.292927846128701e-05,"train/train/tensor_act_model_layers_72_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/max_abs":0.236328125,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/std":5.712356083093023e-05,"train/train/tensor_act_model_layers_31_self_attn_k_proj/norm":5484.2237832451065,"train/train/tensor_param_model_layers_23_mlp_waleed_W_g_weight/std":0.0230712890625,"train/train/tensor_param_model_layers_59_mlp_waleed_W_g_weight/std":0.0294189453125,"train/train/tensor_act_model_layers_60_self_attn_v_proj/mean":-0.001277923583984375,"train/train/tensor_act_model_layers_17_input_layernorm/max_abs":5.25,"train/train/tensor_grad_model_layers_22_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/norm":0.0033812767324342802,"train/train/tensor_act_model_layers_64/norm":15199.232281693952,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_k_proj/mean":-0.0055084228515625,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_8/param/frac_near_user_limit":0,"train/train/layer__model_layers_89/param/mean":0.0016381416975429018,"train/train/tensor_grad_model_layers_67_mlp_waleed_W_u_weight/max_abs":0.00070953369140625,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_o_proj/mean":-0.00035762786865234375,"train/train/tensor_act_model_layers_68_post_attention_layernorm/max_abs":5.125,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/max_abs":0.00225830078125,"train/train/tensor_act_model_layers_40_mlp/std":0.06726109617228557,"train/train/tensor_act_model_layers_11_mlp_waleed_W_g/mean":0.0004906654357910156,"train/train/tensor_act_model_layers_86_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_k_proj/std":0.9511739037588052,"train/train/tensor_act_model_layers_44_mlp_down_proj/max_abs":0.59375,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/norm":0.0028214115777752055,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_91_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/mean":8.58306884765625e-05,"train/train/tensor_param_model_layers_40_mlp_waleed_W_u_weight/std":0.026123046875,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/std":7.631420824010988e-05,"train/train/tensor_act_model_layers_54_mlp_down_proj/std":0.07714846846204856,"train/train/tensor_act_model_layers_21_self_attn_o_proj/max_abs":2.734375,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/std":5.1558473523282295e-05,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/std":0.0247802734375,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/mean":-1.2958480510860682e-08,"train/train/layer_model_layers_31/act/norm":19758.632123350228,"train/train/tensor_act_model_layers_25_self_attn_q_proj/max_abs":6.59375,"train/train/tensor_act_model_layers_29_post_attention_layernorm/max_abs":5.59375,"train/train/tensor_act_model_layers_16/std":2.9102061399701373,"train/train/tensor_act_model_layers_11/norm":17400.32725373717,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/norm":0.025461099095634997,"train/train/layer_model_layers_86/grad/norm":0.06945282526241199,"train/train/tensor_act_model_layers_24_mlp/std":0.08398650451787402,"train/train/tensor_param_model_layers_48_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/norm":4.4375,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/std":0.03857421875,"train/train/tensor_act_model_layers_23_self_attn_q_proj/norm":7576.664183837339,"train/train/tensor_grad_model_layers_5_mlp_waleed_W_u_weight/max_abs":0.0021209716796875,"train/train/tensor_act_model_layers_93/frac_near_user_limit":0,"train/train/layer__model_layers_18/param/norm":19.25817855889414,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/mean":2.09808349609375e-05,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/mean":7.152557373046875e-05,"train/train/tensor_param_model_layers_1_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_45/grad/frac_near_user_limit":0,"train/train/layer__model_layers_25/param/mean":0.0016142418157663806,"train/train/tensor_act_model_layers_16_mlp_waleed/frac_near_user_limit":0,"train/train/layer_model_layers_56/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_18_mlp_waleed_W_g_weight/max_abs":0.00101470947265625,"train/train/tensor_act_model_layers_4_self_attn_k_proj/max_abs":6.65625,"train/train/tensor_act_model_layers_2/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/std":3.722194884983481e-05,"train/train/tensor_act_model_layers_58_self_attn_v_proj/max_abs":3.6875,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/norm":3.5625,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/std":0.00014522727880618202,"train/train/layer_model_layers_84/grad/mean":-8.306641846476776e-08,"train/train/tensor_param_model_layers_1_mlp_waleed_W_g_weight/norm":5.21875,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/std":2.9445815041912176e-05,"train/train/tensor_act_model_layers_24/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_v_proj/max_abs":2.40625,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/max_abs":0.000701904296875,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/norm":7.1875,"train/train/layer_model_layers_61/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_82/grad/mean":-2.621742061305902e-07,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/mean":0.00037384033203125,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/std":3.769944013014518e-05,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/std":4.3355527070984596e-05,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/mean":3.08966264128685e-07,"train/train/tensor_act_model_layers_34_input_layernorm/std":1.0000004239481324,"train/train/tensor_act_model_layers_33_mlp_waleed_W_g/norm":2214.912921530591,"train/train/tensor_act_model_layers_27/std":2.726619020451368,"train/train/tensor_act_model_layers_52/mean":0.006206512451171875,"train/train/layer_model_layers_66/grad/max_abs":0.00091552734375,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/std":0.0286865234375,"train/train/tensor_act_model_layers_44_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/norm":0.00266353297704277,"train/train/tensor_param_model_layers_75_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_87_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_input_layernorm/max_abs":4.9375,"train/train/layer__model_layers_13/param/norm":19.267798184406022,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/norm":0.015106478601855867,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/norm":0.0013414831898429496,"train/train/tensor_param_model_layers_77_mlp_waleed_W_g_weight/max_abs":0.1953125,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/norm":4.1875,"train/train/tensor_act_model_layers_50_self_attn/max_abs":1.375,"train/train/tensor_param_model_layers_52_mlp_waleed_W_g_weight/mean":-9.965896606445312e-05,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_post_attention_layernorm/mean":0.0015273094177246094,"train/train/tensor_act_model_layers_1_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/layer__model_layers_12/param/max_abs":1,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/norm":2.90625,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer_model_layers_28/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_mlp_waleed_W_g_weight/norm":0.015320249756758003,"train/train/tensor_act_model_layers_15_self_attn_v_proj/max_abs":2.125,"train/train/tensor_act_model_layers_31_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/std":0.03759765625,"train/train/tensor_act_model_layers_42_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36/max_abs":25.375,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_waleed_W_g_weight/norm":4.21875,"train/train/tensor_act_model_layers_72/std":2.7578472976327166,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_58_post_attention_layernorm/max_abs":4.90625,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/std":1.4371828476429922e-05,"train/train/tensor_act_model_layers_21_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/std":0.0517578125,"train/train/tensor_act_model_layers_84_self_attn_q_proj/mean":-0.0596923828125,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/max_abs":0.287109375,"train/train/tensor_act_model_layers_8_self_attn/max_abs":1.9453125,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/max_abs":4.71875,"train/train/tensor_act_model_layers_18_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_post_attention_layernorm/norm":5792.612792969337,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/std":0.0238037109375,"train/train/tensor_act_model_layers_89_self_attn_q_proj/max_abs":6.53125,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_down_proj/std":0.06726109617228557,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/mean":0.00013065338134765625,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/mean":5.387701094150543e-07,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/max_abs":0.000675201416015625,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_input_layernorm/std":1.0000000345462463,"train/train/tensor_act_model_layers_55_mlp_waleed_W_u/std":0.3667002145023754,"train/train/tensor_act_model_layers_12_mlp/max_abs":0.5078125,"train/train/tensor_act_model_layers_34_self_attn_v_proj/std":0.39599710535565225,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/norm":5.15625,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/mean":0.00015163421630859375,"train/train/layer_model_layers_40/grad/max_abs":0.00115203857421875,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/mean":-3.771856427192688e-07,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/mean":-6.866455078125e-05,"train/train/tensor_act_model_layers_12_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_waleed_W_u_weight/norm":4.40625,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/max_abs":0.0017547607421875,"train/train/tensor_act_model_layers_67_mlp_waleed_W_u/norm":3508.5456987071507,"train/train/tensor_act_model_layers_20_self_attn_v_proj/max_abs":2.875,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/max_abs":0.00011968612670898438,"train/train/tensor_act_model_layers_83_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp_waleed/norm":1232.650642533621,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_q_proj/std":1.271489320619402,"train/train/tensor_grad_model_layers_16_mlp_waleed_W_g_weight/mean":-5.2852556109428406e-08,"train/train/tensor_act_model_layers_16_mlp_waleed_W_g/mean":-0.0005407333374023438,"train/train/tensor_act_model_layers_10_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp_waleed/mean":-0.0012264251708984375,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/std":1.6912799638104403e-05,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/norm":6,"train/train/tensor_act_model_layers_76_mlp/std":0.18750005419132446,"train/train/tensor_act_model_layers_42_input_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_71/param/max_abs":1,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_mlp_waleed_W_u/max_abs":2.46875,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/mean":1.055002212524414e-05,"train/train/tensor_act_model_layers_52_mlp_down_proj/max_abs":0.90234375,"train/train/tensor_param_model_layers_84_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/mean":0.00015544891357421875,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/mean":-0.00014972686767578125,"train/train/tensor_act_model_layers_68_self_attn_v_proj/mean":-0.01312255859375,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/std":2.3143277156710236e-05,"train/train/tensor_act_model_layers_11_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_v_proj/norm":1959.3000649224043,"train/train/tensor_act_model_layers_43_self_attn_v_proj/max_abs":3.1875,"train/train/tensor_act_model_layers_42_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp/mean":-0.0005955696105957031,"train/train/tensor_act_model_layers_68_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/norm":3.28125,"train/train/tensor_act_model_layers_20_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22/std":2.7539591461624093,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/mean":0.000354766845703125,"train/train/tensor_act_model_layers_12_mlp_waleed_W_g/max_abs":2.078125,"train/train/tensor_act_model_layers_78_mlp/norm":1168.3149186528387,"train/train/tensor_param_model_layers_66_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_6/param/mean":0.0015436133802773986,"train/train/tensor_act_model_layers_52_self_attn_k_proj/std":0.8330100194760923,"train/train/layer__model_layers_34/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_post_attention_layernorm/std":1.0000012818119073,"train/train/tensor_act_model_layers_25_mlp/std":0.07006878776098212,"train/train/tensor_grad_model_layers_56_mlp_waleed_W_g_weight/max_abs":0.000789642333984375,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/norm":0.015278063997805043,"train/train/tensor_act_model_layers_50_mlp/norm":541.6640901350356,"train/train/tensor_act_model_layers_9_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/layer__model_layers_19/param/norm":19.174167039030014,"train/train/layer__model_layers_83/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_42/param/norm":19.86729656134045,"train/train/tensor_param_model_layers_38_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_88/norm":21360.028553175238,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn_o_proj/max_abs":2.75,"train/train/tensor_act_model_layers_66_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/norm":3.140625,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/norm":0.004388957993009085,"train/train/tensor_act_model_layers_50_self_attn_v_proj/max_abs":2.5625,"train/train/layer_model_layers_63/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_input_layernorm/mean":0.0009047985076904297,"train/train/layer_model_layers_7/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_post_attention_layernorm/max_abs":5.46875,"train/train/tensor_act_model_layers_86_self_attn_v_proj/norm":3121.7562207403907,"train/train/tensor_act_model_layers_51_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_56/grad/norm":0.03617334468615418,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_k_proj/mean":-0.0181884765625,"train/train/tensor_act_model_layers_32_post_attention_layernorm/max_abs":5.46875,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/norm":3.84375,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_12_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp/std":0.10132101497846731,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/mean":2.297747414559126e-08,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_q_proj/mean":-0.080322265625,"train/train/tensor_param_model_layers_9_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_58_self_attn_k_proj/mean":0.05810546875,"train/train/tensor_param_model_layers_93_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_12_self_attn_k_proj/std":1.1230521354344771,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/mean":0.000316619873046875,"train/train/tensor_grad_model_layers_9_mlp_waleed_W_g_weight/norm":0.02301308692875729,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn_q_proj/std":0.7734376142157694,"train/train/tensor_grad_model_layers_39_mlp_waleed_W_u_weight/norm":0.014663720106091642,"train/train/tensor_act_model_layers_78_self_attn_k_proj/max_abs":5.375,"train/train/tensor_grad_model_layers_19_mlp_waleed_W_g_weight/mean":-2.2980384528636932e-07,"train/train/tensor_act_model_layers_9_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_91_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_q_proj/max_abs":6.3125,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/mean":-1.2153759598731995e-07,"train/train/tensor_act_model_layers_31_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/mean":-3.3483956940472126e-08,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/std":4.0571086880976414e-05,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/norm":0.002966716330177637,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/norm":4.6875,"train/train/tensor_act_model_layers_72_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/mean":5.671754479408264e-07,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/norm":5.21875,"train/train/tensor_param_model_layers_68_mlp_waleed_W_g_weight/std":0.032470703125,"train/train/tensor_act_model_layers_76_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_waleed_W_g_weight/std":0.0233154296875,"train/train/tensor_act_model_layers_77_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/std":0.037841796875,"train/train/tensor_act_model_layers_66_mlp/mean":0.000576019287109375,"train/train/tensor_param_model_layers_86_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/max_abs":0.166015625,"train/train/tensor_act_model_layers_52_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/std":0.054931640625,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/norm":3.765625,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/std":0.0240478515625,"train/train/tensor_act_model_layers_46_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/mean":2.923421561717987e-06,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38/mean":0.0028352737426757812,"train/train/tensor_act_model_layers_17_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_k_proj/mean":0.01605224609375,"train/train/tensor_act_model_layers_81/frac_near_user_limit":0,"train/train/layer_model_layers_11/act/std":0.9614333766624181,"train/train/tensor_grad_model_layers_21_mlp_waleed_W_g_weight/max_abs":0.0007781982421875,"train/train/tensor_act_model_layers_89_mlp_down_proj/mean":-0.00921630859375,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/max_abs":0.00092315673828125,"train/train/tensor_act_model_layers_65_self_attn_v_proj/mean":-0.009796142578125,"train/train/tensor_param_model_layers_47_mlp_waleed_W_u_weight/norm":4.78125,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/norm":0.03375790432805163,"train/train/tensor_act_model_layers_43_self_attn_v_proj/std":0.38330189042629437,"train/train/tensor_act_model_layers_63/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp_waleed/norm":899.2419354758583,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/max_abs":0.00010156631469726562,"train/train/tensor_act_model_layers_43_self_attn_o_proj/mean":-0.0020961761474609375,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/norm":0.02275514722067765,"train/train/layer_model_layers_21/act/std":0.9187209690913936,"train/train/tensor_act_model_layers_19_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/std":0.037109375,"train/train/tensor_grad_model_layers_12_mlp_waleed_W_g_weight/norm":0.021531919410830068,"train/train/tensor_act_model_layers_44_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/mean":9.202957153320312e-05,"train/train/tensor_grad_model_layers_69_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_post_attention_layernorm/mean":0.0006604194641113281,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_38_self_attn_q_proj/mean":-0.075927734375,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/mean":0.000247955322265625,"train/train/tensor_act_model_layers_88_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/mean":0.0002346038818359375,"train/train/layer_model_layers_12/grad/norm":0.047163814515131804,"train/train/tensor_act_model_layers_44_self_attn_q_proj/max_abs":5.6875,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/norm":4.9375,"train/train/tensor_act_model_layers_5_mlp_waleed_W_g/mean":0.003692626953125,"train/train/layer__model_layers_68/param/std":0.0526007782518863,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_post_attention_layernorm/max_abs":5.4375,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/max_abs":0.000186920166015625,"train/train/tensor_act_model_layers_14/std":2.9414556295798673,"train/train/tensor_act_model_layers_72_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_25/grad/frac_near_dtype_limit":0,"train/train/layer__model_layers_2/param/std":0.046848890562456794,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/max_abs":0.134765625,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/std":0.00026046688954786175,"train/train/tensor_act_model_layers_30_self_attn_v_proj/max_abs":3.9375,"train/train/tensor_act_model_layers_14_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_mlp_waleed_W_u_weight/max_abs":0.000797271728515625,"train/train/tensor_param_model_layers_46_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/max_abs":0.000507354736328125,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/mean":9.802170097827911e-08,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_88_mlp_down_proj/max_abs":4.90625,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/std":8.946776006078181e-06,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/mean":8.678436279296875e-05,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/norm":0.020480819376101952,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/norm":0.03150691283530875,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/norm":7.46875,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/std":9.474746182662405e-05,"train/train/layer__model_layers_91/param/max_abs":1,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/std":0.04296875,"train/train/tensor_act_model_layers_12/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_mlp_waleed_W_u_weight/max_abs":0.138671875,"train/train/tensor_param_model_layers_45_mlp_waleed_W_g_weight/norm":4.78125,"train/train/tensor_act_model_layers_57_mlp_waleed_W_u/mean":0.00580596923828125,"train/train/tensor_act_model_layers_64_self_attn_v_proj/norm":2649.2995270371034,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_51/grad/std":5.32895382364745e-05,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/mean":4.14624810218811e-06,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/std":6.153033307773736e-05,"train/train/tensor_param_model_layers_12_mlp_waleed_W_g_weight/max_abs":0.1328125,"train/train/tensor_act_model_layers_93_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_16/param/max_abs":1,"train/train/tensor_act_model_layers_83_input_layernorm/std":1.0000003617023394,"train/train/tensor_act_model_layers_29_self_attn_v_proj/norm":2355.264269470224,"train/train/tensor_act_model_layers_31_mlp/mean":0.000499725341796875,"train/train/tensor_act_model_layers_80_mlp_waleed_W_u/norm":4633.909813234002,"train/train/tensor_act_model_layers_88_self_attn_k_proj/std":1.121101500473108,"train/train/tensor_param_model_layers_21_mlp_waleed_W_g_weight/std":0.0238037109375,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_mlp_waleed_W_g_weight/mean":-5.2928924560546875e-05,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/norm":7.125,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/norm":2.875,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/std":7.129480707702165e-05,"train/train/tensor_act_model_layers_91/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_81/act/max_abs":25.75,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/std":1.9415219626535406e-05,"train/train/tensor_act_model_layers_37_mlp_down_proj/mean":0.00150299072265625,"train/train/tensor_act_model_layers_75_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_k_proj/max_abs":4.625,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/mean":-1.2009695637971163e-07,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/std":3.9105385706214776e-05,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/norm":4.8125,"train/train/layer__model_layers_84/param/max_abs":1,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_mlp_waleed_W_u_weight/max_abs":0.00077056884765625,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/max_abs":0.330078125,"train/train/tensor_act_model_layers_19_mlp_down_proj/std":0.06396484605342373,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/std":0.0260009765625,"train/train/tensor_act_model_layers_93_self_attn_k_proj/max_abs":6.21875,"train/train/tensor_act_model_layers_27_self_attn_o_proj/max_abs":1.7265625,"train/train/tensor_act_model_layers_36_post_attention_layernorm/std":1.000000694648506,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp/max_abs":0.8203125,"train/train/tensor_param_model_layers_41_mlp_waleed_W_g_weight/std":0.0257568359375,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/mean":-1.3958197087049484e-07,"train/train/tensor_act_model_layers_66_self_attn_o_proj/mean":0.00237274169921875,"train/train/tensor_act_model_layers_86/std":3.488288162243749,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_waleed_W_g_weight/std":4.35194801860573e-05,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/std":2.154921434927166e-05,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/std":0.053955078125,"train/train/tensor_act_model_layers_72_self_attn_o_proj/mean":-0.0059051513671875,"train/train/tensor_act_model_layers_36_input_layernorm/mean":0.004177093505859375,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/max_abs":0.25390625,"train/train/tensor_act_model_layers_45_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_input_layernorm/std":1.0000010731158608,"train/train/tensor_act_model_layers_89_post_attention_layernorm/norm":5792.606201177776,"train/train/tensor_act_model_layers_70_mlp/std":0.12792969125372758,"train/train/tensor_act_model_layers_8_self_attn/mean":-0.0010700225830078125,"train/train/tensor_act_model_layers_60_mlp_waleed_W_g/norm":3076.8029174521434,"train/train/tensor_act_model_layers_91/max_abs":36,"train/train/tensor_act_model_layers_67_self_attn/max_abs":2.5625,"train/train/layer_model_layers_68/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/max_abs":0.00152587890625,"train/train/tensor_act_model_layers_62_self_attn_k_proj/std":0.9228534224153375,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/mean":-0.0003604888916015625,"train/train/tensor_act_model_layers_67_self_attn_v_proj/max_abs":3.75,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/mean":8.367002010345459e-06,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/max_abs":0.0009918212890625,"train/train/tensor_act_model_layers_21_self_attn/norm":1417.8294786161323,"train/train/tensor_act_model_layers_17_mlp_waleed/max_abs":1.96875,"train/train/tensor_act_model_layers_14_self_attn/std":0.07544146981705886,"train/train/tensor_param_model_layers_22_mlp_waleed_W_u_weight/std":0.0240478515625,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/std":0.03662109375,"train/train/tensor_act_model_layers_58_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37/std":2.656274398969844,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_q_proj/std":1.2617250566354452,"train/train/tensor_act_model_layers_49_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_84/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp/max_abs":2.453125,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn_k_proj/max_abs":5.15625,"train/train/tensor_act_model_layers_36_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_69/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_waleed_W_u/mean":-0.0017223358154296875,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/mean":-3.175227902829647e-08,"train/train/layer_model_layers_69/grad/frac_near_dtype_limit":0,"train/train/layer__model_layers_51/param/std":0.050679578246743566,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/mean":-2.2798776626586914e-05,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/max_abs":0.0001010894775390625,"train/train/tensor_grad_model_layers_51_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_waleed_W_u/norm":2788.742580480491,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/norm":0.016975455980539164,"train/train/tensor_act_model_layers_8_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37/max_abs":25.375,"train/train/tensor_act_model_layers_15/std":2.9258308922900564,"train/train/tensor_act_model_layers_77_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/std":1.6924582312175485e-05,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_mlp_waleed_W_g_weight/norm":6.21875,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/max_abs":0.2451171875,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/mean":9.857467375695705e-08,"train/train/tensor_act_model_layers_57_input_layernorm/mean":0.000861048698425293,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/norm":2.859375,"train/train/tensor_grad_model_layers_2_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_7/grad/std":8.454373965776781e-05,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/std":0.0218505859375,"train/train/tensor_act_model_layers_42_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/mean":-6.504706107079983e-08,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/norm":6.0625,"train/train/tensor_act_model_layers_25_self_attn/norm":394.77467145618203,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/mean":-5.5730342864990234e-06,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/max_abs":0.0012359619140625,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/max_abs":0.1318359375,"train/train/tensor_act_model_layers_37_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_mlp_waleed_W_g_weight/mean":-1.125154085457325e-07,"train/train/tensor_act_model_layers_35_mlp_waleed_W_g/std":0.2851563071551331,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_13/param/mean":0.0015764147182708598,"train/train/layer_model_layers_80/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/norm":3.96875,"train/train/tensor_grad_model_layers_70_mlp_waleed_W_g_weight/mean":2.3283064365386963e-07,"train/train/tensor_grad_model_layers_34_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/mean":-0.0001386404037475586,"train/train/tensor_act_model_layers_71_self_attn_q_proj/mean":0.03240966796875,"train/train/tensor_act_model_layers_72_input_layernorm/mean":0.00383758544921875,"train/train/tensor_param_model_layers_17_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_down_proj/norm":460.84734975882213,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/std":6.979234473313598e-05,"train/train/tensor_act_model_layers_4/max_abs":27.25,"train/train/tensor_param_model_layers_57_mlp_waleed_W_u_weight/norm":5.34375,"train/train/tensor_act_model_layers_76_self_attn_v_proj/mean":-0.0023040771484375,"train/train/tensor_param_model_layers_0_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_self_attn_k_proj/norm":6555.628079584568,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/max_abs":0.000377655029296875,"train/train/tensor_act_model_layers_78_self_attn/max_abs":2.984375,"train/train/tensor_act_model_layers_28_self_attn/max_abs":2.09375,"train/train/tensor_act_model_layers_77/std":2.9179778322336394,"train/train/layer__model_layers_70/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/std":0.021484375,"train/train/tensor_act_model_layers_77_mlp_down_proj/norm":1095.5890614664113,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp_waleed/std":0.09582544739805461,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/std":0.041015625,"train/train/layer__model_layers_21/param/norm":19.836022117287023,"train/train/tensor_param_model_layers_9_mlp_waleed_W_g_weight/std":0.0230712890625,"train/train/tensor_act_model_layers_48_self_attn_q_proj/norm":6086.022133919886,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_waleed_W_u_weight/std":6.950914977320042e-05,"train/train/tensor_act_model_layers_18_mlp_waleed_W_g/max_abs":1.96875,"train/train/tensor_grad_model_layers_83_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_input_layernorm/std":1.0000003955987395,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/max_abs":0.24609375,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/mean":0.0001220703125,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp_down_proj/std":0.06616256297908811,"train/train/tensor_act_model_layers_87_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_mlp_waleed_W_g_weight/norm":0.016940857100607613,"train/train/tensor_act_model_layers_16_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_input_layernorm/norm":5792.610961915394,"train/train/tensor_param_model_layers_81_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/mean":0.000171661376953125,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/mean":0.000732421875,"train/train/tensor_act_model_layers_80_mlp_down_proj/mean":0.006103515625,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/norm":0.001624546058975196,"train/train/tensor_act_model_layers_2_post_attention_layernorm/std":1.000000050058587,"train/train/layer_model_layers_26/grad/frac_near_user_limit":0,"train/train/layer__model_layers_36/param/mean":0.0015243911148047485,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/norm":0.0007350843436078149,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/norm":0.010224299603773081,"train/train/tensor_act_model_layers_47_self_attn_k_proj/mean":-0.00201416015625,"train/train/layer_model_layers_51/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_86/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn/mean":-0.00492095947265625,"train/train/tensor_act_model_layers_12_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/norm":0.008088827860514515,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/mean":0.000102996826171875,"train/train/tensor_act_model_layers_52_self_attn_k_proj/mean":0.05572509765625,"train/train/layer_model_layers_17/act/mean":-0.0013742446899414062,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_82_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_mlp_waleed_W_g_weight/mean":0.00012111663818359375,"train/train/tensor_grad_model_layers_10_mlp_waleed_W_g_weight/max_abs":0.0021820068359375,"train/train/tensor_act_model_layers_60_self_attn_o_proj/mean":-5.620718002319336e-05,"train/train/tensor_act_model_layers_32_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_1/grad/max_abs":0.0146484375,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/norm":5.71875,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/std":7.100502622017728e-05,"train/train/tensor_act_model_layers_47_post_attention_layernorm/std":1.0000015470056354,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_k_proj/mean":0.03289794921875,"train/train/tensor_param_model_layers_76_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_input_layernorm/max_abs":5.625,"train/train/tensor_param_model_layers_27_mlp_waleed_W_g_weight/mean":-8.106231689453125e-05,"train/train/tensor_act_model_layers_66_mlp_waleed/mean":-2.5272369384765625e-05,"train/train/tensor_grad_model_layers_82_mlp_waleed_W_u_weight/std":8.095478306667068e-05,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/max_abs":0.00103759765625,"train/train/tensor_act_model_layers_52_post_attention_layernorm/mean":0.0005351901054382324,"train/train/tensor_param_model_layers_62_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/norm":2.90625,"train/train/tensor_param_model_layers_57_mlp_waleed_W_g_weight/std":0.0294189453125,"train/train/tensor_act_model_layers_1/norm":19891.105665239284,"train/train/tensor_act_model_layers_6_mlp_waleed_W_g/std":0.3491221565450484,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_61_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_1/act/std":1.3998739024520872,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/max_abs":0.00093841552734375,"train/train/layer_model_layers_42/act/std":0.8445469135779068,"train/train/tensor_act_model_layers_61_mlp_waleed_W_u/max_abs":2.6875,"train/train/tensor_act_model_layers_9_mlp/norm":544.5867110453496,"train/train/tensor_grad_model_layers_15_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp_waleed_W_u/mean":0.0037841796875,"train/train/tensor_act_model_layers_74_post_attention_layernorm/mean":0.002521514892578125,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/norm":5.46875,"train/train/layer_model_layers_62/grad/std":4.688008913356658e-05,"train/train/tensor_act_model_layers_59_self_attn_k_proj/norm":5605.585358459926,"train/train/layer_model_layers_22/grad/norm":0.04407032373302272,"train/train/tensor_grad_model_layers_72_mlp_waleed_W_g_weight/max_abs":0.0006866455078125,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/std":0.0306396484375,"train/train/tensor_act_model_layers_3_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_k_proj/mean":-0.019134521484375,"train/train/tensor_grad_model_layers_69_mlp_waleed_W_u_weight/norm":0.019166982758721314,"train/train/tensor_param_model_layers_43_mlp_waleed_W_u_weight/std":0.0262451171875,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_37/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/max_abs":0.000827789306640625,"train/train/tensor_param_model_layers_24_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer_model_layers_40/act/mean":-0.0012925788760185242,"train/train/tensor_act_model_layers_4_self_attn_q_proj/max_abs":8.125,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/norm":0.004774757686553278,"train/train/tensor_act_model_layers_51_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/norm":0.028352193317138806,"train/train/tensor_act_model_layers_28_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/max_abs":0.00026702880859375,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/mean":0.0002841949462890625,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_q_proj/max_abs":6,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/mean":-1.2700911611318588e-07,"train/train/tensor_act_model_layers_19_mlp_down_proj/norm":371.2307213999571,"train/train/tensor_act_model_layers_68_self_attn_v_proj/max_abs":3.28125,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/norm":0.019813688892349447,"train/train/layer_model_layers_69/grad/mean":-1.4795679105053453e-07,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/mean":6.780028343200684e-07,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/mean":0.00018978118896484375,"train/train/tensor_act_model_layers_29_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_mlp_waleed_W_u_weight/std":0.0244140625,"train/train/tensor_act_model_layers_62_mlp_waleed/std":0.14550781753398592,"train/train/tensor_act_model_layers_37_mlp_waleed_W_u/mean":-0.00017511844635009766,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/std":9.203768409773641e-05,"train/train/tensor_act_model_layers_82/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/std":0.04931640625,"train/train/tensor_act_model_layers_33_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp_waleed_W_g/norm":2925.518944203051,"train/train/tensor_param_model_layers_80_mlp_waleed_W_u_weight/max_abs":0.2275390625,"train/train/tensor_grad_model_layers_63_mlp_waleed_W_g_weight/norm":0.018130420992235188,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/mean":0.00018596649169921875,"train/train/tensor_act_model_layers_3_self_attn_q_proj/norm":4555.334703129376,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_q_proj/max_abs":5.28125,"train/train/tensor_act_model_layers_78_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_q_proj/max_abs":5.625,"train/train/tensor_grad_model_layers_3_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn/norm":587.135952066428,"train/train/tensor_act_model_layers_9_mlp_down_proj/max_abs":0.69140625,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp_waleed_W_g/norm":2339.7503144644934,"train/train/layer_model_layers_74/grad/max_abs":0.0020599365234375,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/mean":1.8054561223834753e-07,"train/train/tensor_act_model_layers_45_mlp_waleed_W_u/max_abs":2.5,"train/train/tensor_grad_model_layers_17_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_k_proj/std":1.1953126351817684,"train/train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn_v_proj/mean":0.00318145751953125,"train/train/tensor_act_model_layers_45_self_attn_q_proj/std":1.0078125942585037,"train/train/tensor_act_model_layers_61_mlp_down_proj/norm":571.0707887493845,"train/train/tensor_param_model_layers_80_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_75_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_mlp_waleed_W_g_weight/mean":3.1184754334390163e-08,"train/train/tensor_act_model_layers_24_self_attn/max_abs":1.3046875,"train/train/tensor_act_model_layers_84_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/mean":2.684537321329117e-07,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/mean":1.4841556549072266e-05,"train/train/tensor_act_model_layers_12_mlp_waleed_W_g/norm":2036.3721824103784,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_27_self_attn/max_abs":1.7265625,"train/train/tensor_act_model_layers_65/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_waleed_W_u_weight/std":0.0263671875,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp/norm":311.7933821182262,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/std":0.00022017965461344674,"train/train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp/norm":343.8894856265886,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/mean":-6.0558319091796875e-05,"train/train/layer__model_layers_73/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_mlp_waleed_W_g_weight/mean":-2.069864422082901e-07,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_20/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/mean":-0.00011348724365234375,"train/train/tensor_param_model_layers_37_mlp_waleed_W_u_weight/mean":0.0002803802490234375,"train/train/tensor_param_model_layers_76_mlp_waleed_W_u_weight/std":0.036865234375,"train/train/tensor_act_model_layers_45_mlp_waleed_W_u/mean":0.00554656982421875,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/std":0.0439453125,"train/train/layer__model_layers_33/param/norm":19.42997493002114,"train/train/tensor_act_model_layers_4_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/std":2.8942040017823817e-05,"train/train/tensor_act_model_layers_49_self_attn_o_proj/norm":182.37344812187726,"train/train/tensor_param_model_layers_5_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/std":0.000303764422890079,"train/train/tensor_param_model_layers_21_mlp_waleed_W_u_weight/norm":4.3125,"train/train/tensor_act_model_layers_30_self_attn/std":0.14649581924079685,"train/train/layer_model_layers_37/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_waleed_W_g/std":0.2597656483040706,"train/train/global/grad/max_abs":0.056884765625,"train/train/tensor_param_model_layers_29_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_60/norm":15010.021975702524,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_input_layernorm/mean":-0.00572967529296875,"train/train/tensor_act_model_layers_80_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/max_abs":0.2431640625,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_waleed_W_g_weight/mean":2.8137583285570145e-07,"train/train/tensor_param_model_layers_32_mlp_waleed_W_g_weight/max_abs":0.12451171875,"train/train/tensor_act_model_layers_56_self_attn/max_abs":1.8203125,"train/train/tensor_act_model_layers_60_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_57/param/mean":0.001651841281170778,"train/train/tensor_grad_model_layers_27_mlp_waleed_W_g_weight/max_abs":0.000732421875,"train/train/tensor_param_model_layers_83_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_24_mlp_down_proj/max_abs":0.85546875,"train/train/tensor_act_model_layers_29_self_attn/norm":773.8830190003374,"train/train/tensor_param_model_layers_80_mlp_waleed_W_g_weight/norm":7.0625,"train/train/tensor_act_model_layers_29_mlp/norm":271.246752586825,"train/train/tensor_act_model_layers_38_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_waleed_W_g/std":0.4707033295353964,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/mean":3.57162207365036e-07,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/norm":0.014040838883853355,"train/train/tensor_act_model_layers_50_mlp_waleed/max_abs":4.34375,"train/train/tensor_param_model_layers_54_mlp_waleed_W_u_weight/mean":0.00026702880859375,"train/train/tensor_act_model_layers_26_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp_waleed_W_u/mean":-0.0012445449829101562,"train/train/layer_model_layers_78/grad/norm":0.06306972870259135,"train/train/tensor_grad_model_layers_78_mlp_waleed_W_u_weight/max_abs":0.00057220458984375,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_mlp_waleed_W_g_weight/mean":-1.3899989426136017e-07,"train/train/tensor_act_model_layers_71_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/layer__model_layers_50/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_waleed_W_g/std":0.36962989698888893,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/max_abs":0.11376953125,"train/train/tensor_act_model_layers_90_mlp_waleed/std":0.6445372810925994,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/norm":0.003016217227334661,"train/train/layer_model_layers_48/act/norm":20595.09679676428,"train/train/tensor_act_model_layers_55_self_attn_v_proj/mean":0.006134033203125,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/max_abs":0.00095367431640625,"train/train/tensor_act_model_layers_71_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_mlp_waleed_W_g_weight/max_abs":0.00084686279296875,"train/train/tensor_act_model_layers_33_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/std":2.2087609317071388e-05,"train/train/tensor_act_model_layers_83_input_layernorm/norm":5792.606079102503,"train/train/tensor_act_model_layers_2_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp_waleed_W_g/norm":7051.408161381751,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/max_abs":0.00093841552734375,"train/train/tensor_param_model_layers_92_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_77_mlp_waleed_W_u_weight/norm":6.78125,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/std":5.2832535127798124e-05,"train/train/tensor_grad_model_layers_53_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn/mean":0.0005640983581542969,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_waleed_W_u_weight/mean":-0.00022602081298828125,"train/train/tensor_param_model_layers_90_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_36/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_waleed_W_g_weight/mean":-2.7865171432495117e-06,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/std":2.271596501701606e-05,"train/train/tensor_grad_model_layers_13_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_mlp_waleed_W_u_weight/mean":3.0384398996829987e-08,"train/train/layer__model_layers_11/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_21_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_mlp_waleed_W_g_weight/max_abs":0.2138671875,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/norm":0.023561625810995968,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_self_attn_v_proj/mean":0.00388336181640625,"train/train/layer_model_layers_2/act/norm":24125.753057231686,"train/train/tensor_act_model_layers_56/std":2.5859744144456105,"train/train/tensor_act_model_layers_9_self_attn/norm":555.7039763433191,"train/train/tensor_param_model_layers_39_mlp_waleed_W_g_weight/max_abs":0.142578125,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/norm":0.0159926086280882,"train/train/tensor_act_model_layers_79_mlp/norm":1210.9663292434855,"train/train/tensor_act_model_layers_22_mlp_waleed_W_g/norm":2423.376745312994,"train/train/tensor_act_model_layers_64_mlp_waleed/mean":-0.0015087127685546875,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/norm":0.02066371333326627,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/max_abs":0.1953125,"train/train/tensor_act_model_layers_79_input_layernorm/mean":0.00649261474609375,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_waleed_W_u_weight/std":0.0269775390625,"train/train/tensor_grad_model_layers_15_mlp_waleed_W_u_weight/mean":1.2933742254972458e-07,"train/train/tensor_act_model_layers_5_mlp_waleed_W_g/std":0.3261719322846985,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/std":6.392425290709255e-05,"train/train/tensor_act_model_layers_1_mlp_waleed_W_u/max_abs":5.40625,"train/train/tensor_act_model_layers_73_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/norm":0.002894498173013749,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_post_attention_layernorm/std":1.0000010958386518,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp_waleed/mean":-0.0016574859619140625,"train/train/tensor_param_model_layers_17_mlp_waleed_W_g_weight/norm":4.21875,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/norm":4.28125,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/norm":0.02202794219271423,"train/train/tensor_act_model_layers_90_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_mlp_waleed_W_u_weight/max_abs":0.0027008056640625,"train/train/tensor_act_model_layers_66_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/max_abs":0.1611328125,"train/train/tensor_act_model_layers_38_self_attn_k_proj/mean":-0.008636474609375,"train/train/tensor_act_model_layers_28_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/max_abs":0.15625,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_45_mlp_waleed/max_abs":4.6875,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_mlp_waleed_W_g/max_abs":3.953125,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/mean":1.6409903764724731e-06,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp_waleed_W_g/max_abs":2.234375,"train/train/tensor_param_model_layers_70_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_62_self_attn_q_proj/mean":0.021728515625,"train/train/layer__model_layers_62/param/std":0.05165254143077644,"train/train/tensor_param_model_layers_4_mlp_waleed_W_g_weight/norm":4.28125,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/max_abs":0.177734375,"train/train/tensor_act_model_layers_4_mlp_waleed/std":0.2326664036490376,"train/train/tensor_param_model_layers_69_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/std":1.827179969848784e-05,"train/train/tensor_param_model_layers_64_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_92/mean":-0.0467529296875,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/max_abs":0.00010347366333007812,"train/train/tensor_act_model_layers_57_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_82_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_59_mlp_waleed_W_u_weight/max_abs":0.14453125,"train/train/tensor_act_model_layers_18_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/mean":-7.778406143188477e-06,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_o_proj/max_abs":1.4375,"train/train/tensor_act_model_layers_89_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/norm":4.4375,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/std":6.271010035506436e-05,"train/train/tensor_act_model_layers_70_mlp_waleed_W_g/norm":3471.0214252362994,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_mlp_waleed_W_g/norm":2329.544520777737,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_waleed_W_g_weight/norm":0.02498172747890804,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_k_proj/mean":-0.021392822265625,"train/train/tensor_act_model_layers_0_mlp_waleed_W_u/std":1.1230521210257696,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn_q_proj/mean":0.0810546875,"train/train/tensor_act_model_layers_54_mlp_waleed_W_u/mean":-0.000555872917175293,"train/train/tensor_act_model_layers_64_mlp_down_proj/norm":601.7636948557183,"train/train/tensor_act_model_layers_72_post_attention_layernorm/norm":5792.60656738646,"train/train/tensor_act_model_layers_60_mlp_waleed_W_g/std":0.37548923689958724,"train/train/layer__model_layers_93/param/max_abs":1,"train/train/tensor_param_model_norm_weight/mean":1,"train/train/tensor_grad_model_layers_80_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_waleed_W_u/norm":3986.892440134289,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_75/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_44/act/mean":-0.002581283450126648,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/max_abs":0.248046875,"train/train/tensor_grad_model_layers_17_mlp_waleed_W_g_weight/std":4.853065487128298e-05,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_51/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp/mean":-0.00115203857421875,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/max_abs":0.0001773834228515625,"train/train/tensor_act_model_layers_58_mlp_down_proj/norm":476.7418686017982,"train/train/tensor_act_model_layers_27_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/norm":0.0008715873706077068,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/mean":0.0002460479736328125,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp/norm":749.5972536418375,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/norm":0.004889174131897107,"train/train/tensor_act_model_layers_68_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_40/param/max_abs":1,"train/train/tensor_act_model_layers_7/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp_waleed/mean":0.003498077392578125,"train/train/layer_model_layers_64/act/norm":20014.478558217266,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/norm":0.00032411039407812117,"train/train/tensor_act_model_layers_66_mlp/max_abs":0.88671875,"train/train/tensor_act_model_layers_44_mlp_waleed_W_u/norm":2595.3852680757736,"train/train/tensor_grad_model_layers_15_mlp_waleed_W_u_weight/max_abs":0.0020904541015625,"train/train/tensor_act_model_layers_45/max_abs":24.625,"train/train/tensor_act_model_layers_7_self_attn_v_proj/std":0.3657236557604344,"train/train/tensor_act_model_layers_84_self_attn_v_proj/max_abs":4.4375,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/std":0.0238037109375,"train/train/tensor_act_model_layers_14_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/max_abs":0.000530242919921875,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer_model_layers_75/grad/mean":8.618602218242963e-08,"train/train/layer_model_layers_64/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/norm":4.65625,"train/train/tensor_act_model_layers_31_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/max_abs":0.09423828125,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/std":4.052154562821671e-05,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp_waleed_W_g/std":0.3710939068542952,"train/train/tensor_param_model_layers_61_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_53/param/max_abs":1,"train/train/tensor_grad_model_layers_78_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_input_layernorm/std":1.000000063184414,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/std":0.02392578125,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/norm":0.015732662499811693,"train/train/tensor_grad_model_layers_76_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88/mean":-0.00881195068359375,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/max_abs":0.00142669677734375,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/mean":-4.500150680541992e-06,"train/train/layer__model_layers_7/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63/mean":0.014801025390625,"train/train/tensor_act_model_layers_6_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_o_proj/std":0.09131191012946183,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/max_abs":0.263671875,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/std":0.038818359375,"train/train/tensor_act_model_layers_81_input_layernorm/std":1.0000002686865266,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_q_proj/norm":6424.770263762945,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/max_abs":0.0004425048828125,"train/train/tensor_act_model_layers_75_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_post_attention_layernorm/mean":-0.00652313232421875,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/max_abs":0.1728515625,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn_o_proj/mean":-0.0005655288696289062,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_mlp_waleed_W_u_weight/mean":7.677078247070312e-05,"train/train/tensor_act_model_layers_47_mlp_waleed_W_u/mean":-0.001224517822265625,"train/train/tensor_param_model_layers_36_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_72_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/norm":0.02667634353793611,"train/train/tensor_act_model_layers_6_self_attn_v_proj/norm":2342.110436128279,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/mean":-0.00031280517578125,"train/train/tensor_act_model_layers_24_mlp_waleed/std":0.1048597639925305,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/mean":0.00011968612670898438,"train/train/layer__model_layers_4/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_mlp_waleed_W_g/max_abs":3.40625,"train/train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_post_attention_layernorm/max_abs":5.1875,"train/train/tensor_act_model_layers_25_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_mlp_waleed_W_g_weight/mean":1.962762326002121e-07,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/max_abs":0.18359375,"train/train/tensor_grad_model_layers_42_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/mean":0.0002765655517578125,"train/train/tensor_act_model_layers_91_mlp/mean":0.0112762451171875,"train/train/tensor_act_model_layers_2_input_layernorm/std":1.0000000534055276,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/std":3.826834802326195e-05,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/std":0.03857421875,"train/train/tensor_act_model_layers_78_self_attn_o_proj/mean":-0.00492095947265625,"train/train/layer_model_layers_19/grad/mean":-3.564415583837423e-08,"train/train/tensor_param_model_layers_16_mlp_waleed_W_g_weight/mean":-1.537799835205078e-05,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/norm":3.28125,"train/train/tensor_act_model_layers_41_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_mlp_waleed_W_g_weight/max_abs":0.00110626220703125,"train/train/tensor_act_model_layers_3_mlp/std":0.5830233560260079,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/norm":0.01727391971517184,"train/train/layer_model_layers_81/grad/mean":-2.47191318111747e-07,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/std":3.38179095382208e-05,"train/train/tensor_grad_model_layers_10_mlp_waleed_W_u_weight/norm":0.028608435002316784,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_post_attention_layernorm/norm":5792.611938478264,"train/train/tensor_act_model_layers_93_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/layer__model_layers_23/param/norm":20.10497353734891,"train/train/tensor_act_model_layers_75_input_layernorm/mean":0.00335693359375,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_3/act/max_abs":27.25,"train/train/tensor_act_model_layers_55_mlp/mean":-0.00038242340087890625,"train/train/tensor_act_model_layers_84_self_attn_k_proj/mean":0.0887451171875,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/max_abs":0.0002765655517578125,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/max_abs":0.00118255615234375,"train/train/tensor_act_model_layers_81/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp_down_proj/norm":343.8894856265886,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/mean":-2.74099875241518e-07,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/mean":4.752655513584614e-08,"train/train/tensor_act_model_layers_63_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_down_proj/mean":0.0001709461212158203,"train/train/tensor_act_model_layers_93_self_attn/std":0.604501622485357,"train/train/tensor_act_model_layers_57_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/max_abs":0.000865936279296875,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/max_abs":0.150390625,"train/train/tensor_grad_model_layers_69_mlp_waleed_W_u_weight/std":5.305417535499732e-05,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn/std":0.04614261692701119,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_grad_model_layers_76_mlp_waleed_W_g_weight/norm":0.02502694466053728,"train/train/tensor_act_model_layers_73_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/std":0.027587890625,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/max_abs":0.00019550323486328125,"train/train/layer__model_layers_79/param/norm":23.04513234937044,"train/train/tensor_act_model_layers_68/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/mean":-2.437736839056015e-07,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/std":3.761222804667934e-05,"train/train/tensor_act_model_layers_52_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/mean":-8.456408977508545e-06,"train/train/layer_model_layers_27/grad/std":5.431245951556024e-05,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/std":0.00014735892241351292,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/max_abs":0.25390625,"train/train/tensor_act_model_layers_61_input_layernorm/mean":0.004055023193359375,"train/train/tensor_act_model_layers_9_mlp_waleed_W_u/mean":0.007049560546875,"train/train/tensor_param_model_layers_65_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_k_proj/std":1.1601629979486447,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_v_proj/norm":3008.8265404794847,"train/train/tensor_act_model_layers_39/std":2.6406500193361606,"train/train/tensor_act_model_layers_16_mlp_down_proj/norm":336.7701577860633,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/norm":0.03464238471971738,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/mean":-1.698499545454979e-07,"train/train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_post_attention_layernorm/norm":5792.610351567244,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/max_abs":0.236328125,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/norm":0.0223455367007517,"train/train/tensor_act_model_layers_50_mlp_waleed_W_g/mean":0.0070343017578125,"train/train/tensor_act_model_layers_9_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/layer__model_layers_40/param/mean":0.0014812927722187012,"train/train/layer_model_layers_77/act/max_abs":26,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/std":9.067055007406381e-05,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_mlp_waleed_W_g/std":0.30468752471586735,"train/train/tensor_act_model_layers_5_mlp/norm":913.1975908258453,"train/train/layer_model_layers_9/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_post_attention_layernorm/max_abs":5.1875,"train/train/tensor_act_model_layers_69_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/mean":-1.89291313290596e-07,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/max_abs":0.0030517578125,"train/train/tensor_act_model_layers_5_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/norm":0.0008900290418577279,"train/train/tensor_act_model_layers_70_mlp_waleed_W_g/max_abs":2.8125,"train/train/layer_model_layers_2/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_29/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/mean":-3.67872416973114e-07,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/std":6.934442189384401e-06,"train/train/layer__model_layers_23/param/std":0.049613904487742465,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/norm":0.0023313137760655883,"train/train/tensor_act_model_layers_91_input_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_53/param/norm":20.178385374655846,"train/train/tensor_act_model_layers_54_self_attn_v_proj/std":0.3808593954031278,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/mean":0.000644683837890625,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/max_abs":0.00016021728515625,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/norm":3.34375,"train/train/tensor_act_model_layers_81_mlp_down_proj/mean":0.00701141357421875,"train/train/layer_model_layers_48/grad/max_abs":0.0022430419921875,"train/train/tensor_act_model_layers_40_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_54_mlp_waleed_W_g_weight/std":3.999810327144473e-05,"train/train/tensor_act_model_layers_29_self_attn/std":0.13378909220374843,"train/train/tensor_act_model_layers_68_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp/std":0.06396484605342373,"train/train/tensor_act_model_layers_60_mlp_waleed/mean":-0.0005846023559570312,"train/train/tensor_act_model_layers_40_self_attn_v_proj/mean":0.002971649169921875,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/std":0.0263671875,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp/max_abs":0.478515625,"train/train/tensor_act_model_layers_83_mlp/std":0.3051769633270258,"train/train/tensor_grad_model_layers_33_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/norm":0.026970288437229865,"train/train/layer_model_layers_91/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_o_proj/norm":723.4390312814463,"train/train/tensor_act_model_layers_84_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/std":4.360075301457864e-05,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/mean":-8.521601557731628e-08,"train/train/tensor_param_model_layers_6_mlp_waleed_W_u_weight/std":0.023193359375,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/mean":-5.888938903808594e-05,"train/train/tensor_act_model_layers_27_post_attention_layernorm/mean":0.0012912750244140625,"train/train/tensor_act_model_layers_37_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_69/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_13_mlp_waleed_W_g/mean":-0.003726959228515625,"train/train/tensor_act_model_layers_33_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_48/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_q_proj/mean":-0.018829345703125,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/norm":0.0005612533375833566,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_45_self_attn_o_proj/max_abs":2.125,"train/train/tensor_act_model_layers_87_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_q_proj/mean":0.04193115234375,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/max_abs":0.00127410888671875,"train/train/tensor_act_model_layers_57_self_attn_q_proj/std":1.0800838315099035,"train/train/tensor_act_model_layers_87_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_mlp_waleed_W_g_weight/mean":2.1606683731079102e-07,"train/train/layer_model_layers_2/grad/norm":0.16551428121294967,"train/train/tensor_act_model_layers_6_post_attention_layernorm/std":1.0000000952495567,"train/train/tensor_act_model_layers_77_self_attn_v_proj/norm":3089.0552485667667,"train/train/tensor_param_model_layers_56_mlp_waleed_W_u_weight/std":0.0286865234375,"train/train/tensor_act_model_layers_14_self_attn_o_proj/std":0.07544146981705886,"train/train/tensor_act_model_layers_28_self_attn_o_proj/std":0.20092936609984285,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/norm":5.8125,"train/train/tensor_act_model_layers_47_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp/mean":-0.002750396728515625,"train/train/tensor_act_model_layers_26_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/mean":2.2708263713866472e-08,"train/train/tensor_act_model_layers_27_self_attn_k_proj/std":1.0449273595042101,"train/train/tensor_act_model_layers_49_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/std":7.324813560553216e-05,"train/train/tensor_act_model_layers_5_self_attn_o_proj/mean":0.001613616943359375,"train/train/tensor_act_model_layers_52_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/max_abs":0.000789642333984375,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/mean":-1.7578713595867157e-08,"train/train/tensor_act_model_layers_77_input_layernorm/max_abs":5.5,"train/train/tensor_act_model_layers_47_self_attn_q_proj/mean":-0.02691650390625,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/std":2.8552830139278306e-05,"train/train/layer_model_layers_7/grad/norm":0.06852384033664961,"train/train/tensor_act_model_layers_23_self_attn_o_proj/norm":2107.423249407683,"train/train/layer_model_layers_34/grad/std":4.213848205415936e-05,"train/train/tensor_act_model_layers_52_self_attn_q_proj/mean":0.0014314651489257812,"train/train/tensor_act_model_layers_66/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/norm":5.5,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/mean":1.2619420886039734e-07,"train/train/tensor_act_model_layers_26_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/std":4.566608716903637e-05,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_34/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_q_proj/std":1.2578125609929502,"train/train/tensor_act_model_layers_3/max_abs":27.25,"train/train/tensor_act_model_layers_27_mlp_waleed_W_g/norm":2263.2290587921184,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_54/grad/mean":7.843959047902207e-08,"train/train/tensor_param_model_layers_79_mlp_waleed_W_g_weight/norm":6.90625,"train/train/tensor_param_model_layers_28_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_16_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/mean":-7.528811693191528e-06,"train/train/tensor_act_model_layers_75_mlp_down_proj/mean":0.003726959228515625,"train/train/tensor_act_model_layers_26_self_attn_o_proj/mean":-0.000213623046875,"train/train/tensor_act_model_layers_3_input_layernorm/mean":-0.00809478759765625,"train/train/tensor_param_model_layers_73_mlp_waleed_W_u_weight/std":0.035400390625,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/std":2.961050583376623e-05,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/mean":-0.0001010894775390625,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/max_abs":0.0026397705078125,"train/train/layer__model_layers_85/param/norm":24.75288019857487,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/std":7.440118408476677e-05,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/std":7.511896970136644e-05,"train/train/tensor_act_model_layers_49_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_self_attn_k_proj/max_abs":4.15625,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/norm":0.0006636531791978136,"train/train/tensor_act_model_layers_66_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_q_proj/norm":5737.875697644819,"train/train/tensor_grad_model_norm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_k_proj/mean":0.0245361328125,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_o_proj/mean":-0.0004891157150268555,"train/train/tensor_param_model_layers_22_mlp_waleed_W_g_weight/mean":7.2479248046875e-05,"train/train/tensor_param_model_layers_30_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp_down_proj/mean":-0.0003304481506347656,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/max_abs":0.1630859375,"train/train/tensor_act_model_layers_27/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_o_proj/max_abs":1.9140625,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/mean":-1.979060471057892e-09,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/std":2.9162662409389628e-05,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/mean":8.950519259087741e-08,"train/train/tensor_param_model_layers_55_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn_v_proj/max_abs":3.671875,"train/train/tensor_grad_model_layers_9_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_61_mlp/max_abs":0.7890625,"train/train/tensor_act_model_layers_23_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_o_proj/norm":760.2183491231991,"train/train/tensor_act_model_layers_82_post_attention_layernorm/norm":5792.607543948074,"train/train/tensor_grad_model_layers_35_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/max_abs":0.000598907470703125,"train/train/tensor_act_model_layers_34_self_attn_v_proj/norm":2292.9924622148183,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/norm":5.6875,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/norm":0.009059249854169229,"train/train/layer_model_layers_56/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/std":3.177129746827654e-05,"train/train/tensor_act_model_layers_5_self_attn/norm":1289.0682275198708,"train/train/tensor_param_model_layers_36_mlp_waleed_W_u_weight/norm":4.5625,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/std":5.442537128378315e-05,"train/train/tensor_param_model_layers_86_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/std":0.03466796875,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/max_abs":0.19921875,"train/train/tensor_act_model_layers_15_self_attn_v_proj/mean":-0.00276947021484375,"train/train/tensor_act_model_layers_7_mlp/std":0.1472190254589959,"train/train/tensor_act_model_layers_62/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/mean":-6.798654794692993e-08,"train/train/tensor_act_model_layers_89_post_attention_layernorm/std":1.0000001582720746,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/max_abs":0.109375,"train/train/tensor_act_model_layers_76_self_attn_v_proj/std":0.46484377817446354,"train/train/tensor_act_model_layers_17_self_attn_o_proj/max_abs":1.25,"train/train/layer_model_layers_26/act/norm":19944.649709237798,"train/train/tensor_act_model_layers_69_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/mean":-0.0002384185791015625,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_mlp_waleed_W_u_weight/norm":4.15625,"train/train/tensor_act_model_layers_37_mlp_waleed_W_g/max_abs":2.515625,"train/train/tensor_param_model_layers_38_mlp_waleed_W_g_weight/max_abs":0.12255859375,"train/train/tensor_act_model_layers_77_post_attention_layernorm/std":1.0000006319894785,"train/train/tensor_act_model_layers_50_self_attn_k_proj/std":0.9433614185856074,"train/train/tensor_act_model_layers_54_mlp_down_proj/max_abs":0.66796875,"train/train/tensor_param_model_layers_53_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_69/std":2.7031492445112066,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/norm":5.40625,"train/train/tensor_act_model_layers_24_post_attention_layernorm/mean":0.0018768310546875,"train/train/tensor_act_model_layers_0_mlp_down_proj/norm":13490.920508464298,"train/train/tensor_act_model_layers_33_mlp_waleed_W_u/mean":0.0038299560546875,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/max_abs":0.00012493133544921875,"train/train/tensor_act_model_layers_82_self_attn_k_proj/max_abs":5.5,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/norm":0.03214583255415936,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/max_abs":0.00018024444580078125,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/max_abs":0.00015735626220703125,"train/train/tensor_grad_model_layers_3_mlp_waleed_W_g_weight/norm":0.059169106119755815,"train/train/tensor_act_model_layers_33_post_attention_layernorm/std":1.000000405931391,"train/train/tensor_param_model_layers_15_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_47_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_68_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer_model_layers_32/act/mean":0.0013364925980567932,"train/train/layer_model_layers_50/grad/std":4.797179032423865e-05,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/norm":0.005891236463224033,"train/train/tensor_grad_model_layers_52_mlp_waleed_W_u_weight/norm":0.015550046194796007,"train/train/tensor_act_model_layers_36_mlp_waleed_W_g/mean":-0.00585174560546875,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/norm":0.0011887358125677126,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/std":3.273031853100977e-05,"train/train/layer__model_layers_30/param/norm":19.767459154030142,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/max_abs":0.00115203857421875,"train/train/tensor_act_model_layers_30_self_attn_q_proj/std":0.9873062696453665,"train/train/tensor_act_model_layers_36_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_post_attention_layernorm/std":1.0000007096093424,"train/train/tensor_act_model_layers_65_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/std":0.033935546875,"train/train/tensor_act_model_layers_78_self_attn_v_proj/mean":-0.002292633056640625,"train/train/tensor_act_model_layers_90_self_attn_v_proj/max_abs":8.1875,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/mean":-4.4563785195350647e-07,"train/train/tensor_grad_model_layers_17_mlp_waleed_W_g_weight/norm":0.017552993611201752,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/std":0.030029296875,"train/train/tensor_act_model_layers_74_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/max_abs":0.10791015625,"train/train/tensor_param_model_layers_7_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_mlp_waleed_W_g_weight/std":0.023681640625,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/mean":2.1827872842550278e-09,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/norm":2.75,"train/train/tensor_act_model_layers_48_self_attn_q_proj/max_abs":9.3125,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/std":0.0537109375,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/std":0.00016114635212179616,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/mean":-0.00020885467529296875,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/std":1.1577274573519341e-05,"train/train/tensor_act_model_layers_0_self_attn_k_proj/std":0.8037127224507216,"train/train/tensor_act_model_layers_34/std":2.671898683852918,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/mean":-2.461019903421402e-07,"train/train/tensor_act_model_layers_36_mlp_waleed_W_g/max_abs":2.71875,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/norm":0.0238782449384537,"train/train/tensor_param_model_layers_56_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/norm":0.000694929589761579,"train/train/tensor_act_model_layers_74_mlp/norm":971.9932275607126,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/mean":1.2922100722789764e-08,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/norm":0.010676137564415107,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/std":3.707661743757637e-05,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_2_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp_down_proj/mean":-0.00018227100372314453,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/max_abs":0.000675201416015625,"train/train/tensor_act_model_layers_83_mlp_waleed/norm":2627.6807264985773,"train/train/tensor_grad_model_layers_67_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/std":6.19953750373758e-05,"train/train/tensor_param_model_layers_31_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_mlp_waleed_W_g_weight/mean":-6.76955096423626e-08,"train/train/tensor_act_model_layers_42_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/norm":0.027228384517963958,"train/train/tensor_act_model_layers_5_mlp_down_proj/max_abs":0.9375,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_mlp/mean":-0.0088958740234375,"train/train/tensor_act_model_layers_26_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_mlp_waleed_W_u/std":0.30468751763990576,"train/train/tensor_act_model_layers_69_post_attention_layernorm/max_abs":5.0625,"train/train/tensor_grad_model_layers_87_mlp_waleed_W_u_weight/max_abs":0.00109100341796875,"train/train/tensor_act_model_layers_66_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn_v_proj/norm":1712.7511031077138,"train/train/tensor_act_model_layers_13_self_attn_v_proj/mean":0.00823974609375,"train/train/tensor_param_model_layers_73_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_66_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/mean":-0.0001125335693359375,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/norm":4.71875,"train/train/tensor_act_model_layers_36_self_attn_v_proj/max_abs":3.75,"train/train/tensor_grad_model_layers_57_mlp_waleed_W_u_weight/norm":0.016389388547795074,"train/train/tensor_param_model_layers_24_mlp_waleed_W_g_weight/max_abs":0.12890625,"train/train/tensor_act_model_layers_67_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_v_proj/frac_near_user_limit":0,"train/train/global/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77/norm":16890.76592319725,"train/train/tensor_act_model_layers_76/norm":16679.464805475356,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/max_abs":0.00119781494140625,"train/train/tensor_act_model_layers_39_mlp_down_proj/max_abs":0.63671875,"train/train/tensor_act_model_layers_0_self_attn_o_proj/max_abs":3.15625,"train/train/tensor_act_model_layers_19_mlp_down_proj/mean":0.0007839202880859375,"train/train/tensor_act_model_layers_83_post_attention_layernorm/norm":5792.613159180123,"train/train/tensor_act_model_layers_49_mlp_waleed/max_abs":3,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/norm":0.0007766680539354487,"train/train/tensor_act_model_layers_46_self_attn/mean":0.0003108978271484375,"train/train/layer_model_layers_78/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_waleed_W_u_weight/norm":4.1875,"train/train/layer_model_layers_21/grad/std":6.624953704838197e-05,"train/train/tensor_act_model_layers_69_mlp_down_proj/std":0.1250000466097758,"train/train/tensor_act_model_layers_38_self_attn_k_proj/norm":6016.784335310807,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_69/grad/max_abs":0.00106048583984375,"train/train/layer_model_layers_76/act/norm":21674.389183642103,"train/train/layer_model_layers_70/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/norm":3.4375,"train/train/tensor_act_model_layers_85_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/mean":2.300739288330078e-05,"train/train/tensor_param_model_layers_82_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/norm":4.65625,"train/train/tensor_act_model_layers_59_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_q_proj/std":0.8251970573271413,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_50_mlp/mean":0.002796173095703125,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/mean":6.344635039567947e-09,"train/train/tensor_act_model_layers_61_mlp_waleed_W_u/norm":3144.6706599580875,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_input_layernorm/norm":5792.610107422765,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/max_abs":0.0005950927734375,"train/train/tensor_act_model_layers_85/mean":0.011962890625,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/max_abs":0.001220703125,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/norm":0.04778394008171863,"train/train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/max_abs":0.1845703125,"train/train/layer_model_layers_25/act/max_abs":26.5,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/std":2.4587115921170168e-05,"train/train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/norm":3.109375,"train/train/tensor_param_model_layers_16_mlp_waleed_W_u_weight/max_abs":0.1142578125,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/std":5.704953837868702e-05,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_waleed/std":1.3496138162719258,"train/train/tensor_act_model_layers_3_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn/std":0.17212814355224587,"train/train/tensor_act_model_layers_23_mlp_waleed_W_u/max_abs":2.1875,"train/train/tensor_act_model_layers_92_input_layernorm/mean":-0.0011625289916992188,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/std":0.03759765625,"train/train/tensor_param_model_layers_63_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_45/grad/std":5.0391843581158614e-05,"train/train/tensor_grad_model_layers_71_mlp_waleed_W_u_weight/max_abs":0.00055694580078125,"train/train/tensor_act_model_layers_29_mlp_waleed/std":0.07824735991986757,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/mean":6.323680281639099e-07,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_k_proj/norm":5981.988994980389,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/std":0.00031179930504366706,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/std":1.2796485085341586e-05,"train/train/tensor_act_model_layers_8_mlp/max_abs":0.74609375,"train/train/tensor_act_model_layers_80_mlp_down_proj/norm":1627.3210229970473,"train/train/tensor_grad_model_layers_51_mlp_waleed_W_u_weight/max_abs":0.0004329681396484375,"train/train/tensor_param_model_layers_25_mlp_waleed_W_g_weight/norm":4.34375,"train/train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_mlp_waleed_W_u_weight/mean":1.8719583749771118e-07,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/mean":5.471520125865936e-07,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/mean":0.00020885467529296875,"train/train/tensor_param_model_layers_24_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/max_abs":0.0012054443359375,"train/train/layer_model_layers_26/grad/norm":0.03755874306592493,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/mean":-7.373455446213484e-08,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/mean":-0.0004730224609375,"train/train/tensor_act_model_layers_19_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/mean":4.01865690946579e-07,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_input_layernorm/std":1.0000008025669065,"train/train/tensor_param_model_layers_70_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/max_abs":0.000743865966796875,"train/train/tensor_act_model_layers_58_mlp/mean":0.0005483627319335938,"train/train/tensor_act_model_layers_39_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_down_proj/mean":0.0007200241088867188,"train/train/tensor_act_model_layers_2/norm":19728.98107899047,"train/train/tensor_act_model_layers_50_mlp/max_abs":0.96484375,"train/train/tensor_param_model_layers_78_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_mlp_waleed_W_g_weight/mean":3.119930624961853e-08,"train/train/tensor_act_model_layers_55/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_mlp_waleed_W_g_weight/mean":-0.00010585784912109375,"train/train/tensor_param_model_layers_29_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_26_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_59/param/std":0.05119405074347958,"train/train/tensor_grad_model_layers_89_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp_waleed/mean":-0.0012187957763671875,"train/train/tensor_act_model_layers_77_mlp_down_proj/mean":8.52346420288086e-05,"train/train/tensor_act_model_layers_67_post_attention_layernorm/mean":0.0063934326171875,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/max_abs":0.00067138671875,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/max_abs":0.000858306884765625,"train/train/tensor_act_model_layers_62_self_attn/norm":507.8259041262101,"train/train/tensor_act_model_layers_36_mlp_waleed/max_abs":2.953125,"train/train/tensor_grad_model_layers_72_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_mlp_waleed_W_u_weight/norm":0.0455379533617906,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_waleed_W_g_weight/std":8.678640155006271e-05,"train/train/tensor_act_model_layers_19_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/norm":5.5,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/mean":-6.29425048828125e-05,"train/train/layer_model_layers_34/act/mean":0.002118527889251709,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_mlp_waleed_W_u/max_abs":2.109375,"train/train/tensor_act_model_layers_1_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_waleed_W_g/norm":3027.1955794995233,"train/train/tensor_act_model_layers_35_input_layernorm/norm":5792.606445313128,"train/train/tensor_act_model_layers_14_mlp/max_abs":0.51953125,"train/train/tensor_act_model_layers_6_self_attn/norm":998.7894548358049,"train/train/tensor_param_model_layers_12_mlp_waleed_W_g_weight/std":0.0230712890625,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/mean":0.00017833709716796875,"train/train/tensor_act_model_layers_56_self_attn_o_proj/mean":0.0007982254028320312,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/std":9.121760004213743e-05,"train/train/tensor_act_model_layers_42_self_attn_q_proj/max_abs":7.15625,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/std":6.0984577271268675e-06,"train/train/tensor_act_model_layers_39_mlp_down_proj/std":0.06335485280031684,"train/train/tensor_act_model_layers_71_post_attention_layernorm/std":1.0000010495054032,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/mean":-7.949769496917725e-06,"train/train/tensor_act_model_layers_29_self_attn_o_proj/max_abs":1.1015625,"train/train/tensor_param_model_layers_38_mlp_waleed_W_g_weight/std":0.0255126953125,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/max_abs":0.00019359588623046875,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/std":7.380016793214851e-05,"train/train/tensor_act_model_layers_5_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/std":6.271325780433189e-05,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/max_abs":0.000934600830078125,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/std":7.757738511674026e-05,"train/train/tensor_grad_model_layers_18_mlp_waleed_W_u_weight/norm":0.016179497806573562,"train/train/tensor_act_model_layers_78_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp/mean":0.0009050369262695312,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/mean":-0.0003490447998046875,"train/train/tensor_grad_model_layers_59_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/mean":1.3245153240859509e-07,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/norm":0.013005537744896954,"train/train/tensor_param_model_layers_30_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_10_mlp_waleed_W_g_weight/std":0.023193359375,"train/train/tensor_act_model_layers_12_mlp/std":0.07067903167743803,"train/train/tensor_act_model_layers_67_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_waleed_W_g_weight/norm":0.04431438174106017,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/mean":1,"train/train/layer__model_layers_77/param/max_abs":1,"train/train/tensor_act_model_layers_90_self_attn_k_proj/std":1.2343755975552815,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/max_abs":0.19921875,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/norm":0.0060012838721059366,"train/train/tensor_param_model_layers_33_mlp_waleed_W_g_weight/max_abs":0.12060546875,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_84_mlp_waleed_W_g/std":0.6679713398342231,"train/train/tensor_act_model_layers_49_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_mlp_waleed_W_u_weight/mean":-2.6100315153598785e-07,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_post_attention_layernorm/max_abs":4.59375,"train/train/tensor_param_model_layers_78_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_89_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_mlp_waleed_W_u/norm":3968.608309420824,"train/train/tensor_grad_model_layers_53_mlp_waleed_W_u_weight/std":4.148664400030868e-05,"train/train/tensor_grad_model_layers_48_mlp_waleed_W_g_weight/max_abs":0.000530242919921875,"train/train/tensor_act_model_layers_85_post_attention_layernorm/mean":0.00616455078125,"train/train/tensor_act_model_layers_71_self_attn/max_abs":1.9140625,"train/train/tensor_act_model_layers_47_self_attn_k_proj/max_abs":3.765625,"train/train/tensor_act_model_layers_43_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/mean":-5.50062395632267e-08,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/mean":9.30667738430202e-08,"train/train/tensor_act_model_layers_79_mlp_waleed/norm":2096.509957627652,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/max_abs":0.000537872314453125,"train/train/tensor_act_model_layers_60_mlp_waleed_W_u/mean":-0.00323486328125,"train/train/tensor_act_model_layers_16_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_waleed_W_g/max_abs":3.3125,"train/train/tensor_act_model_layers_42_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/std":0.0294189453125,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/mean":9.298324584960938e-05,"train/train/tensor_act_model_layers_12_mlp_waleed/norm":802.218265114231,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_o_proj/mean":0.0002872943878173828,"train/train/tensor_act_model_layers_64_mlp/mean":0.0005345344543457031,"train/train/tensor_grad_model_layers_74_mlp_waleed_W_u_weight/norm":0.02200696174512832,"train/train/tensor_param_model_layers_34_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer_model_layers_28/act/max_abs":26.375,"train/train/layer_model_layers_82/act/std":1.0609764557020622,"train/train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn/mean":-0.00415802001953125,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/mean":4.2438507080078125e-05,"train/train/layer_model_layers_39/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_v_proj/max_abs":4.5625,"train/train/tensor_param_model_layers_23_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_84_self_attn_k_proj/norm":8747.23216907047,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/std":1.872974186869372e-05,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/norm":4.75,"train/train/layer__model_layers_46/param/norm":19.784514195529923,"train/train/tensor_act_model_layers_37_mlp_down_proj/norm":355.5914642876377,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/max_abs":0.000759124755859375,"train/train/tensor_act_model_layers_77_mlp_waleed_W_g/mean":-0.0144805908203125,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/mean":7.258495315909386e-08,"train/train/tensor_act_model_layers_29_mlp_waleed_W_g/norm":2119.7097916508233,"train/train/layer__model_layers_43/param/std":0.04906528039535977,"train/train/tensor_param_model_layers_13_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp_waleed_W_g/max_abs":2.171875,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/max_abs":0.001373291015625,"train/train/tensor_act_model_layers_20/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/std":3.8730153791265364e-05,"train/train/tensor_act_model_layers_56_self_attn_k_proj/max_abs":5.71875,"train/train/layer_model_layers_41/act/max_abs":25,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/norm":5.8125,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_o_proj/norm":3239.3336442728105,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/norm":3.09375,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_mlp_waleed_W_g/max_abs":2.8125,"train/train/tensor_act_model_layers_9_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/max_abs":0.1982421875,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/mean":8.07642936706543e-06,"train/train/tensor_grad_model_layers_30_mlp_waleed_W_u_weight/norm":0.014589381382980882,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_mlp_waleed_W_u_weight/max_abs":0.0006561279296875,"train/train/tensor_act_model_layers_46_mlp/mean":-0.0008440017700195312,"train/train/tensor_grad_model_layers_9_mlp_waleed_W_u_weight/norm":0.021760809399201796,"train/train/tensor_param_model_layers_55_mlp_waleed_W_g_weight/max_abs":0.150390625,"train/train/tensor_param_model_layers_69_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_waleed_W_g_weight/norm":6.59375,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_47/param/mean":0.0016105520929821568,"train/train/tensor_act_model_layers_44_input_layernorm/std":1.0000013129802503,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/max_abs":0.09814453125,"train/train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_mlp_down_proj/max_abs":0.97265625,"train/train/tensor_act_model_layers_40_post_attention_layernorm/norm":5792.294189454571,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/max_abs":0.0003452301025390625,"train/train/layer_model_layers_9/act/max_abs":26.5,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/mean":1.7881393432617188e-05,"train/train/tensor_act_model_layers_43_self_attn_o_proj/std":0.11597121699279217,"train/train/layer__model_layers_64/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/std":1.7072241656665224e-05,"train/train/tensor_act_model_layers_76_post_attention_layernorm/mean":0.00530242919921875,"train/train/tensor_param_model_layers_46_mlp_waleed_W_u_weight/max_abs":0.1455078125,"train/train/tensor_act_model_layers_1_post_attention_layernorm/max_abs":4.25,"train/train/tensor_param_model_layers_9_mlp_waleed_W_g_weight/max_abs":0.1083984375,"train/train/tensor_act_model_layers_15_self_attn_o_proj/mean":-0.00043392181396484375,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/std":0.0003452301025390625,"train/train/tensor_param_model_layers_34_mlp_waleed_W_u_weight/max_abs":0.1279296875,"train/train/tensor_act_model_layers_80_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/norm":7.625,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/mean":0.0003414154052734375,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_post_attention_layernorm/std":1.0000000816071373,"train/train/tensor_param_model_layers_9_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_45/param/mean":0.0015298394070772596,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/max_abs":0.00045013427734375,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/norm":5.15625,"train/train/layer_model_layers_25/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_mlp_waleed_W_u_weight/norm":0.014940559762151234,"train/train/tensor_act_model_layers_48_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/std":5.4847068996820524e-05,"train/train/tensor_param_model_layers_92_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_waleed_W_u_weight/max_abs":0.000659942626953125,"train/train/tensor_act_model_layers_73_self_attn_v_proj/max_abs":3.421875,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/mean":-4.3353065848350525e-07,"train/train/tensor_act_model_layers_26_self_attn_v_proj/mean":0.0027618408203125,"train/train/layer_model_layers_78/act/mean":0.0038753747940063477,"train/train/tensor_grad_model_layers_43_mlp_waleed_W_u_weight/norm":0.014496781168669174,"train/train/tensor_param_model_layers_64_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_60_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_74/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/mean":8.649658411741257e-08,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/max_abs":0.2216796875,"train/train/tensor_act_model_layers_32_mlp/std":0.0721438687962344,"train/train/tensor_grad_model_layers_33_mlp_waleed_W_u_weight/max_abs":0.000576019287109375,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_v_proj/mean":-0.00518798828125,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/max_abs":0.000568389892578125,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/max_abs":0.10791015625,"train/train/tensor_param_model_layers_6_mlp_waleed_W_g_weight/max_abs":0.10595703125,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/std":6.29129243288298e-06,"train/train/tensor_act_model_layers_54/max_abs":23.75,"train/train/tensor_param_model_layers_12_mlp_waleed_W_g_weight/norm":4.15625,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_59/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/mean":1.3709068298339844e-05,"train/train/tensor_act_model_layers_66_self_attn/norm":1309.424119383697,"train/train/tensor_act_model_layers_33_mlp_waleed_W_g/std":0.270020862920109,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/norm":7.34375,"train/train/tensor_act_model_layers_27_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/norm":3.8125,"train/train/layer_model_layers_19/act/frac_near_user_limit":0,"train/train/layer__model_layers_89/param/max_abs":1,"train/train/tensor_grad_model_layers_58_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_79/param/max_abs":1,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/mean":-1.2956559658050537e-05,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/max_abs":0.01129150390625,"train/train/tensor_act_model_layers_28_self_attn_v_proj/norm":2796.599491619804,"train/train/tensor_act_model_layers_5/std":3.3008254548466525,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/norm":0.02302716592423983,"train/train/tensor_act_model_layers_37_mlp_down_proj/std":0.06140153284789958,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/std":8.050403871621012e-05,"train/train/tensor_act_model_layers_59_self_attn_q_proj/mean":-0.0496826171875,"train/train/tensor_act_model_layers_48_post_attention_layernorm/mean":0.0008754730224609375,"train/train/tensor_act_model_layers_23_mlp_down_proj/frac_near_user_limit":0,"train/train/layer__model_layers_37/param/max_abs":1,"train/train/tensor_act_model_layers_89_mlp_down_proj/norm":3399.882020861135,"train/train/layer__model_layers_84/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_mlp_waleed_W_g/std":0.45117208561593053,"train/train/tensor_act_model_layers_38_self_attn_o_proj/max_abs":2.46875,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp_down_proj/std":0.07067903167743803,"train/train/layer_model_layers_75/act/std":0.934041219162204,"train/train/tensor_act_model_layers_42_mlp_down_proj/max_abs":0.73046875,"train/train/tensor_act_model_layers_75_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/std":0.0245361328125,"train/train/tensor_act_model_layers_59/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp_waleed_W_g/norm":2186.7792436829754,"train/train/tensor_grad_model_layers_68_mlp_waleed_W_g_weight/max_abs":0.000469207763671875,"train/train/tensor_act_model_layers_89_input_layernorm/norm":5792.613647465459,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/max_abs":0.0003814697265625,"train/train/layer_model_layers_66/act/mean":0.011027991771697998,"train/train/layer__model_layers_68/param/norm":21.310764454627737,"train/train/tensor_grad_model_layers_77_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn/mean":-0.002796173095703125,"train/train/tensor_grad_model_layers_29_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/mean":7.62939453125e-05,"train/train/tensor_act_model_layers_23_self_attn_k_proj/std":1.234375543232086,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/mean":2.5667250156402588e-05,"train/train/tensor_param_model_layers_73_input_layernorm_weight/mean":1,"train/train/layer_model_layers_57/grad/mean":-8.980361468706414e-08,"train/train/tensor_param_model_layers_37_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/std":1.2704634015573548e-05,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/max_abs":0.000202178955078125,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/norm":0.001228231816872982,"train/train/tensor_act_model_layers_59_post_attention_layernorm/max_abs":4.90625,"train/train/tensor_grad_model_layers_52_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/norm":0.0023967108747709916,"train/train/layer_model_layers_42/grad/frac_near_user_limit":0,"train/train/layer__model_layers_85/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/std":0.035888671875,"train/train/tensor_act_model_layers_28_mlp_waleed/norm":750.6275433460178,"train/train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/std":0.00013394711980482276,"train/train/tensor_param_model_layers_74_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/std":0.004480397312240112,"train/train/layer_model_layers_79/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/norm":5,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/max_abs":0.00133514404296875,"train/train/layer_model_layers_14/act/max_abs":26.125,"train/train/tensor_act_model_layers_31_mlp_waleed/max_abs":3.109375,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/norm":0.002814109179640718,"train/train/layer_model_layers_0/act/norm":31439.146213168257,"train/train/tensor_act_model_layers_16/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/max_abs":0.00051116943359375,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/max_abs":0.0003204345703125,"train/train/tensor_act_model_layers_34_self_attn_o_proj/std":0.17505070969498088,"train/train/tensor_act_model_layers_69_mlp/std":0.1250000466097758,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_k_proj/mean":-0.01568603515625,"train/train/tensor_act_model_layers_38/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp_waleed_W_g/norm":2426.7924722132766,"train/train/layer_model_layers_34/grad/norm":0.03413842096408533,"train/train/tensor_param_model_layers_14_mlp_waleed_W_u_weight/norm":4.15625,"train/train/tensor_act_model_layers_56_mlp_waleed_W_u/mean":0.0014209747314453125,"train/train/tensor_grad_model_layers_57_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/std":3.129721736635739e-05,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_k_proj/std":1.2187505074035372,"train/train/tensor_grad_model_layers_54_mlp_waleed_W_g_weight/norm":0.014481772564501064,"train/train/tensor_act_model_layers_49_self_attn_q_proj/norm":4394.26920760926,"train/train/tensor_param_model_layers_24_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp_waleed_W_g/max_abs":4.625,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/norm":6.96875,"train/train/tensor_act_model_layers_74_self_attn/mean":0.00588226318359375,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/max_abs":0.18359375,"train/train/tensor_act_model_layers_79_self_attn_q_proj/mean":-0.03564453125,"train/train/tensor_act_model_layers_62_input_layernorm/norm":5792.610595705455,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp_down_proj/std":0.06713912660271366,"train/train/tensor_act_model_layers_82/max_abs":27.875,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/mean":7.772445678710938e-05,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/norm":0.02305166227114448,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_77_mlp_waleed_W_g_weight/mean":1.9406434148550034e-07,"train/train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_44/param/max_abs":1,"train/train/tensor_act_model_layers_71_self_attn_k_proj/norm":4935.914002682039,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn_o_proj/mean":-0.0002110600471496582,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/norm":0.0006658094844318107,"train/train/tensor_grad_model_layers_90_mlp_waleed_W_g_weight/mean":-3.550667315721512e-07,"train/train/tensor_act_model_layers_61_self_attn_v_proj/max_abs":2.09375,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/norm":0.01931278113743859,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_mlp_waleed_W_u_weight/max_abs":0.00098419189453125,"train/train/tensor_act_model_layers_61_self_attn/std":0.12475659434384297,"train/train/tensor_param_model_layers_90_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_12_mlp_waleed_W_u/mean":-0.0003075599670410156,"train/train/tensor_grad_model_layers_41_mlp_waleed_W_u_weight/mean":1.2118835002183914e-07,"train/train/tensor_grad_model_layers_77_mlp_waleed_W_u_weight/std":6.641017398800679e-05,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_mlp_waleed_W_u_weight/max_abs":0.000789642333984375,"train/train/layer_model_layers_30/act/max_abs":26.375,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_63/param/std":0.05188787435033496,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/max_abs":0.1328125,"train/train/tensor_act_model_layers_0_mlp/mean":-0.020477294921875,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_35_mlp_waleed_W_u/mean":-0.00229644775390625,"train/train/tensor_act_model_layers_83_mlp_waleed/max_abs":6.25,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/std":0.00010445971824548373,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/std":0.028076171875,"train/train/tensor_grad_model_layers_20_mlp_waleed_W_g_weight/max_abs":0.00069427490234375,"train/train/layer_model_layers_78/grad/std":7.783969945036103e-05,"train/train/tensor_act_model_layers_47_input_layernorm/max_abs":5.03125,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/mean":-2.971501089632511e-08,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/mean":-2.2794120013713837e-07,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/mean":2.6337802410125732e-06,"train/train/tensor_act_model_layers_76_mlp_waleed_W_g/std":0.5009794402458123,"train/train/tensor_act_model_layers_4_input_layernorm/norm":5792.609985356948,"train/train/tensor_param_model_layers_90_mlp_waleed_W_g_weight/mean":-2.467632293701172e-05,"train/train/tensor_act_model_layers_91_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_mlp_down_proj/std":0.07873568217698812,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_83/grad/std":8.344640657583227e-05,"train/train/tensor_param_model_layers_61_mlp_waleed_W_g_weight/mean":2.3245811462402344e-05,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_mlp_waleed_W_u_weight/norm":5.1875,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/mean":-1.7834827303886414e-07,"train/train/tensor_act_model_layers_20_input_layernorm/std":1.000000015221303,"train/train/tensor_act_model_layers_36_mlp/std":0.06243905876614992,"train/train/tensor_act_model_layers_47_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/max_abs":0.000659942626953125,"train/train/layer__model_layers_17/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_waleed_W_u_weight/std":0.00014765562237695093,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/max_abs":0.119140625,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/mean":2.941815182566643e-07,"train/train/tensor_act_model_layers_53_mlp_waleed/frac_near_dtype_limit":0,"train/train/layer_model_layers_24/grad/norm":0.04864823261655646,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/max_abs":0.0008697509765625,"train/train/tensor_act_model_layers_45_mlp_waleed_W_g/norm":2620.129223857966,"train/train/tensor_act_model_layers_30/std":2.6875234384063944,"train/train/tensor_act_model_layers_77_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_waleed/std":0.07788124278700631,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/max_abs":0.00038909912109375,"train/train/tensor_act_model_layers_26_mlp_waleed/mean":-0.002838134765625,"train/train/tensor_act_model_layers_0_self_attn_q_proj/std":0.46240317834015465,"train/train/tensor_act_model_layers_93_mlp_waleed_W_u/norm":9922.801475726686,"train/train/tensor_act_model_layers_66_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/norm":6.40625,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_waleed_W_g/mean":0.003749847412109375,"train/train/tensor_act_model_layers_5_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/layer_model_layers_76/grad/std":7.785981052952518e-05,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_76/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_q_proj/norm":5726.370429620496,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/std":1.4728448826107524e-05,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/mean":5.471520125865936e-08,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/std":0.06201171875,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/mean":-4.704270395450294e-08,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/norm":0.0014420262072178738,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/max_abs":0.2236328125,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/std":3.838971263977013e-05,"train/train/tensor_act_model_layers_82_input_layernorm/max_abs":5.21875,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/std":0.0380859375,"train/train/tensor_grad_model_layers_48_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_mlp/norm":362.3351169526012,"train/train/layer__model_layers_86/param/max_abs":1,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/norm":0.0006602788079926969,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp_waleed/std":0.12963936348803184,"train/train/tensor_act_model_layers_55_mlp_waleed/norm":1108.883240378225,"train/train/tensor_act_model_layers_91_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/std":0.00016066341060681088,"train/train/tensor_act_model_layers_53_mlp_waleed_W_g/std":0.35742206553938743,"train/train/tensor_act_model_layers_16_input_layernorm/std":1.000000013358658,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/mean":0.00031280517578125,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_20_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/std":0.03857421875,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/std":3.007215667029327e-05,"train/train/layer_model_layers_52/act/mean":0.0039406828582286835,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/std":0.023681640625,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_77/param/std":0.05693524919156396,"train/train/tensor_param_model_layers_90_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_13_mlp_waleed_W_g_weight/mean":-0.00010967254638671875,"train/train/tensor_act_model_layers_85_mlp_waleed_W_g/norm":5338.328575172254,"train/train/layer_model_layers_47/grad/max_abs":0.000888824462890625,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/max_abs":0.0010833740234375,"train/train/tensor_param_model_layers_18_mlp_waleed_W_g_weight/norm":4.25,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/std":4.1400291444711975e-05,"train/train/tensor_param_model_layers_26_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_waleed_W_u/mean":0.03753662109375,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/max_abs":0.00016880035400390625,"train/train/tensor_act_model_layers_52_mlp_waleed_W_u/mean":0.002597808837890625,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_21/grad/norm":0.05366624400456639,"train/train/tensor_act_model_layers_82_mlp_down_proj/mean":-0.0088958740234375,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/std":5.5734081878990376e-05,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/norm":4.3125,"train/train/tensor_act_model_layers_88_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/layer_model_layers_62/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_waleed_W_u_weight/norm":0.053370838190636864,"train/train/tensor_act_model_layers_32_mlp_down_proj/max_abs":0.64453125,"train/train/tensor_act_model_layers_43_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/mean":-0.00032806396484375,"train/train/tensor_act_model_layers_68_self_attn_k_proj/max_abs":5.46875,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/std":2.0246735839004918e-05,"train/train/tensor_act_model_layers_21_input_layernorm/norm":5792.611450198311,"train/train/tensor_act_model_layers_64_mlp_waleed_W_g/mean":-0.00476837158203125,"train/train/tensor_act_model_layers_34_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/mean":-0.00010967254638671875,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/std":5.341246719390601e-05,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_waleed_W_u_weight/norm":0.031468639843320816,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/std":0.054931640625,"train/train/layer__model_layers_68/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/max_abs":0.1357421875,"train/train/tensor_act_model_layers_46_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/layer_model_layers_0/act/std":1.3567570163777722,"train/train/tensor_grad_model_layers_88_mlp_waleed_W_u_weight/norm":0.031132955706373505,"train/train/tensor_param_model_layers_14_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_input_layernorm/mean":0.004734039306640625,"train/train/tensor_act_model_layers_10_post_attention_layernorm/norm":5792.612060550934,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/std":5.439767008833987e-05,"train/train/layer_model_layers_79/act/mean":0.00020706653594970703,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/std":0.0242919921875,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/mean":1.6763806343078613e-07,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/norm":0.002281763010933871,"train/train/tensor_act_model_layers_88_mlp_waleed/max_abs":10.875,"train/train/tensor_act_model_layers_58_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/norm":0.015408627104223867,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn_q_proj/max_abs":6.6875,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_86/act/max_abs":31.875,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_q_proj/norm":6421.634932723644,"train/train/tensor_param_model_layers_19_mlp_waleed_W_g_weight/norm":4.1875,"train/train/tensor_act_model_layers_58_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/max_abs":0.0123291015625,"train/train/tensor_grad_model_layers_18_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/estimated_remaining_minutes":0,"train/train/tensor_act_model_layers_89_mlp_waleed_W_u/mean":-0.019622802734375,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/norm":0.010453996724614627,"train/train/tensor_act_model_layers_65_self_attn/mean":-0.0006933212280273438,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/mean":-3.8853613659739494e-08,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_mlp_waleed/max_abs":3.1875,"train/train/tensor_act_model_layers_78_mlp_down_proj/mean":0.002819061279296875,"train/train/tensor_act_model_layers_31_post_attention_layernorm/std":1.0000003446191645,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_post_attention_layernorm/max_abs":5.34375,"train/train/tensor_act_model_layers_66_mlp_down_proj/max_abs":0.88671875,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/max_abs":0.00010633468627929688,"train/train/tensor_act_model_layers_26_mlp/max_abs":0.53515625,"train/train/tensor_grad_model_layers_85_mlp_waleed_W_g_weight/mean":-4.046596586704254e-07,"train/train/tensor_act_model_layers_18_mlp_waleed/mean":0.0009622573852539062,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/norm":0.017609145026748867,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/max_abs":0.0223388671875,"train/train/tensor_act_model_layers_52_self_attn/std":0.0559139736434932,"train/train/tensor_act_model_layers_61_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp_waleed_W_u/std":0.715824691191808,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/max_abs":0.00012683868408203125,"train/train/tensor_act_model_layers_37_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_input_layernorm/mean":0.005039215087890625,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/std":7.168012927458617e-05,"train/train/layer__model_layers_61/param/norm":20.973203001464153,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_27/param/std":0.0483860556592698,"train/train/tensor_act_model_layers_40_input_layernorm/mean":0.0028276443481445312,"train/train/tensor_act_model_layers_76_input_layernorm/norm":5792.611816420008,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/mean":1.1583324521780014e-07,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/max_abs":0.00043487548828125,"train/train/tensor_param_model_layers_44_mlp_waleed_W_u_weight/mean":-6.771087646484375e-05,"train/train/tensor_param_model_layers_60_mlp_waleed_W_u_weight/mean":-6.628036499023438e-05,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_v_proj/norm":2975.4567878521325,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/std":1.5249388802683829e-05,"train/train/tensor_act_model_layers_91_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/std":4.634828871978882e-05,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/std":2.1638739062308604e-05,"train/train/tensor_act_model_layers_47_post_attention_layernorm/norm":5792.606323244054,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_q_proj/norm":5862.823028239526,"train/train/tensor_act_model_layers_40_mlp_waleed_W_g/max_abs":2.390625,"train/train/tensor_act_model_layers_58_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/mean":0.00012159347534179688,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp_waleed/max_abs":3.3125,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/mean":6.441026926040649e-06,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_post_attention_layernorm/std":1.0000013210560978,"train/train/tensor_act_model_layers_63_input_layernorm/max_abs":4.9375,"train/train/tensor_act_model_layers_83_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61/mean":0.0124664306640625,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp_waleed/norm":759.9924895472595,"train/train/tensor_param_model_layers_10_mlp_waleed_W_g_weight/norm":4.21875,"train/train/tensor_act_model_layers_28_mlp/max_abs":0.9296875,"train/train/layer_model_layers_46/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/norm":3.984375,"train/train/tensor_act_model_rotary_emb/norm":2297.209228515625,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/max_abs":0.1064453125,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/max_abs":0.283203125,"train/train/layer__model_layers_67/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_mlp_waleed/std":0.15673905196540733,"train/train/tensor_act_model_layers_50_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/norm":0.005183161565878662,"train/train/tensor_act_model_layers_85_mlp_down_proj/mean":-0.01092529296875,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_waleed/norm":1019.8015218004136,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/norm":6.40625,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/mean":-2.03610397875309e-07,"train/train/tensor_act_model_layers_54_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp_waleed_W_u/norm":2143.2098147591287,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/norm":0.016361748422099334,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/std":0.00011092566556807113,"train/train/tensor_act_model_layers_64_self_attn_v_proj/max_abs":4.34375,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/mean":2.092123031616211e-05,"train/train/tensor_act_model_layers_86_mlp_waleed_W_u/mean":0.010986328125,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/mean":-1.405179500579834e-05,"train/train/tensor_act_model_layers_76_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_mlp_waleed_W_u_weight/norm":4.59375,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/max_abs":0.189453125,"train/train/tensor_act_model_layers_24_mlp_waleed_W_u/std":0.2617187735749704,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/std":0.046630859375,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_66_mlp_waleed_W_u/mean":0.0147705078125,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_q_proj/std":1.0371149353912779,"train/train/layer_model_layers_11/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_act_model_layers_15_mlp/std":0.06616256297908811,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/std":4.067406032259515e-05,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/norm":5.25,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/mean":-1.0960502550005913e-07,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_down_proj/std":0.12792969125372758,"train/train/tensor_act_model_layers_24_mlp/mean":0.0014057159423828125,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62/std":2.5937754165886324,"train/train/layer__model_layers_32/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_mlp_waleed_W_g_weight/norm":0.01152082025580098,"train/train/tensor_act_model_layers_57_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_post_attention_layernorm/max_abs":5.15625,"train/train/tensor_act_model_layers_85_post_attention_layernorm/max_abs":5.59375,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/norm":0.002241289660997246,"train/train/tensor_act_model_layers_68_mlp_down_proj/std":0.11413597347699378,"train/train/layer_model_layers_32/grad/norm":0.029325652858596184,"train/train/tensor_act_model_layers_27_self_attn_q_proj/norm":6634.191685779115,"train/train/layer_model_layers_71/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_mlp_waleed_W_g_weight/mean":2.6877387426793575e-08,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/max_abs":0.000942230224609375,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/norm":0.038273082562537426,"train/train/layer_model_layers_87/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_o_proj/max_abs":0.84765625,"train/train/layer_model_layers_43/act/mean":-0.006869196891784668,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/mean":-5.6417775340378284e-08,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/mean":-0.00018978118896484375,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_8_mlp_waleed_W_g/norm":2274.813598303601,"train/train/tensor_act_model_layers_91_self_attn_q_proj/norm":7092.540137344382,"train/train/tensor_act_model_layers_47/max_abs":24.5,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/mean":-8.046627044677734e-06,"train/train/tensor_act_model_layers_73_self_attn_o_proj/std":0.3291031018482552,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/mean":-3.695022314786911e-07,"train/train/layer_model_layers_91/act/mean":0.0035071372985839844,"train/train/tensor_param_model_layers_84_input_layernorm_weight/mean":1,"train/train/layer__model_layers_0/param/norm":19.55768835594138,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/std":0.0264892578125,"train/train/layer__model_layers_1/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp_waleed_W_g/std":0.5156250791906347,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/std":0.00026204012393822525,"train/train/tensor_act_model_layers_32_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/norm":6.9375,"train/train/layer_model_layers_45/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_down_proj/norm":307.68916356707376,"train/train/tensor_act_model_layers_24/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_k_proj/mean":-0.0090484619140625,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp/std":0.06750523494245766,"train/train/tensor_act_model_layers_32_mlp/norm":417.60883952066683,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/std":9.799643472772362e-05,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp_waleed_W_g/mean":0.0179443359375,"train/train/tensor_act_model_layers_92_mlp/mean":-0.0302734375,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/std":0.0003452301025390625,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/mean":-1.3485550880432129e-05,"train/train/tensor_act_model_layers_57_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_66/act/norm":20004.428201970153,"train/train/tensor_act_model_layers_4_self_attn_o_proj/mean":0.0006647109985351562,"train/train/tensor_param_model_layers_32_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/max_abs":0.1044921875,"train/train/tensor_act_model_layers_53_mlp_down_proj/max_abs":0.74609375,"train/train/tensor_act_model_layers_80/norm":18162.099881808514,"train/train/tensor_act_model_layers_6_mlp_down_proj/mean":-0.00334930419921875,"train/train/tensor_act_model_layers_33_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/mean":-1.1995434761047363e-06,"train/train/tensor_act_model_layers_6/max_abs":26.875,"train/train/layer_model_layers_15/act/norm":21156.461762552364,"train/train/tensor_act_model_layers_87_mlp/std":0.542002016914864,"train/train/tensor_param_model_layers_46_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/std":0.8789064041773343,"train/train/layer_model_layers_41/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_30/act/norm":19809.396964794963,"train/train/tensor_grad_model_layers_8_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_input_layernorm/max_abs":4.9375,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/max_abs":0.00113677978515625,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/max_abs":0.00015735626220703125,"train/train/tensor_act_model_layers_61/max_abs":24.5,"train/train/tensor_act_model_layers_19/norm":16329.694599518221,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/norm":0.050619454492201656,"train/train/tensor_grad_model_layers_92_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/std":9.394467389313564e-05,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/max_abs":0.228515625,"train/train/tensor_grad_model_layers_60_mlp_waleed_W_u_weight/norm":0.016053290387549785,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/max_abs":0.00052642822265625,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/std":1.5691370089973827e-05,"train/train/tensor_act_model_layers_85_mlp/std":0.4179757175577836,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_mlp_waleed_W_u_weight/std":0.023193359375,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67/mean":0.0149383544921875,"train/train/tensor_param_model_layers_34_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn_v_proj/mean":0.0299072265625,"train/train/tensor_act_model_layers_89_self_attn/std":0.39844903955417277,"train/train/layer_model_layers_76/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/mean":-4.1604042053222656e-05,"train/train/layer_model_layers_28/grad/norm":0.04093821145843232,"train/train/tensor_grad_model_layers_41_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_58/grad/norm":0.04288123992446671,"train/train/tensor_param_model_layers_38_mlp_waleed_W_u_weight/std":0.025390625,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/std":0.05615234375,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/max_abs":0.00055694580078125,"train/train/tensor_param_model_layers_53_mlp_waleed_W_g_weight/mean":-0.00010824203491210938,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/std":0.025634765625,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_33_self_attn_q_proj/std":0.8916032400460145,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/norm":0.002679101839654414,"train/train/tensor_act_model_layers_25_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/std":9.269734797886744e-05,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_33/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp_waleed/std":0.08654811896331946,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/mean":-3.0724331736564636e-06,"train/train/tensor_act_model_layers_42_mlp_waleed/frac_near_user_limit":0,"train/train/layer_model_layers_83/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/max_abs":0.2412109375,"train/train/tensor_grad_model_layers_12_mlp_waleed_W_g_weight/max_abs":0.0028076171875,"train/train/tensor_act_model_layers_8_self_attn_q_proj/mean":0.0662841796875,"train/train/tensor_act_model_layers_10_self_attn/std":0.16870317250845782,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/max_abs":0.1416015625,"train/train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_k_proj/mean":0.00670623779296875,"train/train/tensor_param_model_layers_12_mlp_waleed_W_u_weight/std":0.02294921875,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/norm":2.890625,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/std":8.310427601243315e-05,"train/train/tensor_param_model_layers_22_mlp_waleed_W_u_weight/max_abs":0.1201171875,"train/train/tensor_param_model_layers_48_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp/norm":1755.3542312210461,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/norm":0.02277077634780422,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/norm":0.04321944192482799,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/norm":0.0029260593041534264,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_k_proj/std":0.8925805222903765,"train/train/layer_model_layers_49/act/std":0.8187295415222742,"train/train/tensor_act_model_layers_37_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp_waleed_W_g/norm":2990.0801701011387,"train/train/tensor_act_model_layers_40_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/norm":6.625,"train/train/tensor_act_model_layers_41_post_attention_layernorm/max_abs":5.21875,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/max_abs":0.318359375,"train/train/tensor_param_model_layers_60_mlp_waleed_W_u_weight/max_abs":0.146484375,"train/train/tensor_act_model_layers_28/max_abs":26.375,"train/train/layer_model_layers_1/grad/norm":0.5171262405885242,"train/train/tensor_act_model_layers_73_post_attention_layernorm/std":1.000000802961968,"train/train/layer_model_layers_4/act/max_abs":27.25,"train/train/tensor_act_model_layers_76_self_attn_k_proj/std":1.0390625605849824,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/mean":5.91278076171875e-05,"train/train/tensor_act_model_layers_15_self_attn/std":0.09155442719583311,"train/train/tensor_grad_model_layers_51_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_waleed_W_g/std":0.3046879351900123,"train/train/tensor_act_model_layers_9_self_attn/max_abs":0.84765625,"train/train/tensor_grad_model_layers_43_mlp_waleed_W_g_weight/std":3.683751198875932e-05,"train/train/tensor_act_model_layers_50_self_attn_v_proj/std":0.3896497046008159,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/mean":-0.0003643035888671875,"train/train/tensor_act_model_layers_30_self_attn_k_proj/max_abs":6.03125,"train/train/tensor_act_model_layers_51_mlp_waleed_W_u/std":0.35351607741762936,"train/train/tensor_act_model_layers_75_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_10/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/max_abs":0.00164031982421875,"train/train/tensor_grad_model_layers_27_mlp_waleed_W_u_weight/max_abs":0.000972747802734375,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/max_abs":0.1142578125,"train/train/tensor_grad_model_layers_34_mlp_waleed_W_g_weight/std":3.690243459371445e-05,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/norm":0.001368817354163378,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/max_abs":0.0004367828369140625,"train/train/tensor_act_model_layers_89_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_20/act/mean":-0.0029953867197036743,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_mlp_waleed_W_u/std":0.4277346030280522,"train/train/tensor_act_model_layers_1_mlp_waleed/std":0.9970909040867262,"train/train/tensor_act_model_layers_41_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_42/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/norm":0.014438588332568263,"train/train/tensor_param_model_layers_22_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/std":0.026123046875,"train/train/tensor_act_model_layers_87_self_attn_o_proj/max_abs":4.9375,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_mlp_waleed_W_g_weight/max_abs":0.000858306884765625,"train/train/tensor_grad_model_layers_83_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_waleed_W_g/max_abs":2.453125,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/mean":-1.0848045349121094e-05,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer__model_layers_2/param/max_abs":1,"train/train/tensor_act_model_layers_9_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_waleed/norm":840.9727386621133,"train/train/tensor_grad_model_layers_28_mlp_waleed_W_u_weight/max_abs":0.00092315673828125,"train/train/tensor_grad_model_layers_75_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn/std":0.5586146459868395,"train/train/tensor_param_model_layers_51_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_77/act/mean":0.0009538382291793823,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/std":9.316290108739359e-06,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn_o_proj/mean":0.0011072158813476562,"train/train/tensor_act_model_layers_47_mlp_waleed_W_u/std":0.3085937653491388,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/max_abs":0.095703125,"train/train/tensor_act_model_layers_54_input_layernorm/std":1.000001189827091,"train/train/tensor_act_model_layers_24_self_attn/norm":1274.8418644276617,"train/train/tensor_act_model_layers_48_mlp_waleed_W_u/max_abs":2.890625,"train/train/tensor_act_model_layers_28_self_attn_v_proj/max_abs":2.90625,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/std":6.652266467096514e-05,"train/train/tensor_act_model_layers_18_self_attn_k_proj/norm":5650.911634290062,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_mlp_waleed_W_u_weight/mean":1.658918336033821e-07,"train/train/tensor_act_model_layers_91_mlp/max_abs":9.25,"train/train/tensor_act_model_layers_93_self_attn_v_proj/max_abs":4.71875,"train/train/tensor_param_model_layers_42_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_22/param/norm":19.52356098650423,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/norm":0.05161409397708043,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/std":0.031005859375,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/std":0.00011159756388220668,"train/train/tensor_act_model_layers_8_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_53/act/norm":19227.6429687593,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/max_abs":0.000335693359375,"train/train/tensor_param_model_layers_10_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_89/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_mlp/norm":1085.293398856482,"train/train/tensor_param_model_layers_3_mlp_waleed_W_g_weight/norm":4.375,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/max_abs":0.00049591064453125,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/std":0.03857421875,"train/train/tensor_act_model_layers_13_input_layernorm/max_abs":5.125,"train/train/tensor_act_model_layers_73_input_layernorm/std":1.0000008197722587,"train/train/tensor_act_model_layers_16_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/std":0.0223388671875,"train/train/layer__model_layers_15/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_0/param/std":0.04825000760984998,"train/train/tensor_act_model_layers_53_self_attn_q_proj/std":0.9111344704403364,"train/train/tensor_act_model_layers_15_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_k_proj/max_abs":5.28125,"train/train/tensor_act_model_layers_44/max_abs":24.625,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70/max_abs":25.25,"train/train/tensor_act_model_layers_87/max_abs":32,"train/train/layer_model_layers_44/grad/std":3.820498199775514e-05,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/mean":8.106231689453125e-05,"train/train/tensor_act_model_layers_37_post_attention_layernorm/std":1.0000007973785845,"train/train/tensor_act_model_layers_32_mlp_waleed_W_g/std":0.2846692598061705,"train/train/tensor_grad_model_layers_71_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/mean":-0.001007080078125,"train/train/layer_model_layers_50/grad/mean":7.768370815539695e-08,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/mean":-0.00010919570922851562,"train/train/tensor_act_model_layers_91/std":4.375000453306272,"train/train/tensor_act_model_layers_50_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/max_abs":0.1337890625,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/mean":-1.862645149230957e-07,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/max_abs":0.00024318695068359375,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/mean":3.816094249486923e-07,"train/train/tensor_act_model_layers_32_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/max_abs":0.000934600830078125,"train/train/tensor_act_model_layers_58/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/max_abs":0.00025177001953125,"train/train/tensor_act_model_layers_20_self_attn_v_proj/mean":-0.00048089027404785156,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/mean":-0.0003070831298828125,"train/train/tensor_act_model_layers_28/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/max_abs":0.0002593994140625,"train/train/tensor_act_model_layers_64_mlp/max_abs":0.98828125,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/max_abs":0.000492095947265625,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/max_abs":0.23828125,"train/train/tensor_act_model_layers_51_self_attn_q_proj/mean":0.034912109375,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/std":8.452641422948719e-05,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/norm":0.016449286010285804,"train/train/tensor_act_model_layers_11_post_attention_layernorm/norm":5792.610839852249,"train/train/tensor_act_model_layers_54_self_attn_k_proj/std":0.8125000825295039,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_o_proj/std":0.3168956951040336,"train/train/tensor_act_model_layers_20_self_attn/frac_near_dtype_limit":0,"train/train/epoch_time_elapsed":3872.9421130530536,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/mean":3.618188202381134e-07,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_mlp_down_proj/mean":0.0002567768096923828,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn/max_abs":1.015625,"train/train/tensor_act_model_layers_20_self_attn_k_proj/max_abs":4.65625,"train/train/tensor_act_model_layers_12_mlp/mean":3.059208393096924e-05,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_1/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_27_self_attn_v_proj/std":0.40478607023478763,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_post_attention_layernorm/norm":5792.607788089049,"train/train/tensor_act_model_layers_77_mlp/mean":8.52346420288086e-05,"train/train/tensor_act_model_layers_70_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn_q_proj/mean":0.090576171875,"train/train/tensor_grad_model_layers_26_mlp_waleed_W_u_weight/max_abs":0.0014801025390625,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/max_abs":0.0024261474609375,"train/train/tensor_act_model_layers_63_self_attn_o_proj/std":0.17163560907229375,"train/train/tensor_act_model_layers_16_self_attn_q_proj/mean":-0.04254150390625,"train/train/tensor_act_model_layers_93_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/std":0.00014674763011792574,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/max_abs":0.1259765625,"train/train/tensor_act_model_layers_64_mlp/std":0.10388207598753144,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/max_abs":0.00014019012451171875,"train/train/tensor_param_model_layers_84_mlp_waleed_W_u_weight/norm":7.53125,"train/train/tensor_act_model_layers_10_mlp_waleed_W_u/mean":0.0008325576782226562,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/mean":9.870529174804688e-05,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/std":0.024169921875,"train/train/tensor_act_model_layers_85_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_input_layernorm/std":1.0000000764921395,"train/train/tensor_act_model_layers_65/std":2.6367294381949353,"train/train/tensor_act_model_layers_29_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_mlp_down_proj/mean":0.001857757568359375,"train/train/tensor_act_model_layers_31_mlp_waleed/norm":759.8501722104835,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn_k_proj/max_abs":4.59375,"train/train/tensor_act_model_layers_81_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/mean":-1.932494342327118e-07,"train/train/tensor_act_model_layers_53_mlp_waleed_W_g/max_abs":2.46875,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/mean":1.3760291039943695e-07,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_26_mlp_waleed_W_u_weight/norm":0.01629741536869382,"train/train/tensor_param_model_layers_24_mlp_waleed_W_u_weight/std":0.023681640625,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/max_abs":0.0003337860107421875,"train/train/tensor_act_model_layers_13_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/std":0.027587890625,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/std":6.712193332790634e-05,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/max_abs":0.000972747802734375,"train/train/tensor_act_model_layers_83_self_attn_o_proj/max_abs":3.296875,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/max_abs":0.1455078125,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/norm":2.859375,"train/train/tensor_act_model_layers_0/norm":14716.23331212197,"train/train/tensor_act_model_layers_15_mlp_waleed_W_g/mean":-0.00262451171875,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/max_abs":0.1640625,"train/train/tensor_grad_model_layers_6_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/mean":-6.116926670074463e-06,"train/train/tensor_grad_model_layers_17_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_24/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/std":3.088147086347219e-05,"train/train/layer__model_layers_42/param/std":0.04901707439421403,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/mean":0.00043487548828125,"train/train/layer_model_layers_4/act/norm":24788.09191056926,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/std":4.24442344696077e-05,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp_waleed/norm":1112.6056474274064,"train/train/tensor_act_model_layers_77_self_attn/mean":0.0089569091796875,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/norm":0.03772387435477484,"train/train/tensor_act_model_layers_50_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2/mean":-0.0092315673828125,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/mean":0.0002727508544921875,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_50_self_attn_k_proj/max_abs":5.5,"train/train/layer__model_layers_79/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/mean":-0.000213623046875,"train/train/tensor_grad_model_layers_15_mlp_waleed_W_g_weight/max_abs":0.0011444091796875,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/norm":4.875,"train/train/tensor_param_model_layers_16_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_56/act/max_abs":24,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_mlp_waleed_W_u_weight/max_abs":0.1279296875,"train/train/tensor_act_model_layers_17_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_q_proj/std":1.1328132103227164,"train/train/layer_model_layers_39/grad/max_abs":0.0009918212890625,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_73/act/std":0.9107098703657482,"train/train/tensor_act_model_layers_24_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/grad/norm":0.03782601094542216,"train/train/tensor_act_model_layers_80_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_80/act/std":1.0477499891093212,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp/std":0.05993715720573151,"train/train/layer_model_layers_0/grad/max_abs":0.056884765625,"train/train/tensor_act_model_layers_73_mlp_waleed_W_u/std":0.4619151643655228,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/norm":0.0038682836005142908,"train/train/tensor_act_model_layers_53_input_layernorm/std":1.0000014453038162,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/mean":-2.8014183044433594e-06,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn/max_abs":0.8515625,"train/train/layer_model_layers_25/act/std":0.8744960029969158,"train/train/tensor_grad_model_layers_7_mlp_waleed_W_g_weight/mean":3.925379132851958e-09,"train/train/tensor_act_model_layers_1_mlp_down_proj/std":2.2852197720917644,"train/train/tensor_act_model_layers_74_self_attn_v_proj/std":0.5810573133516093,"train/train/tensor_act_model_layers_80/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_input_layernorm/std":1.0000010449748225,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/norm":0.0020768215314966123,"train/train/tensor_act_model_layers_74_mlp/mean":0.00220489501953125,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn_o_proj/mean":-0.0001175999641418457,"train/train/tensor_act_model_layers_78_self_attn_v_proj/max_abs":3.234375,"train/train/tensor_param_model_layers_5_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp_waleed_W_g/max_abs":2.171875,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/mean":9.059906005859375e-05,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/norm":0.0055710672640301275,"train/train/tensor_grad_model_layers_82_mlp_waleed_W_g_weight/mean":9.703217074275017e-08,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/norm":0.035531059351856666,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/norm":0.004049045538504998,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/max_abs":0.1201171875,"train/train/tensor_act_model_layers_43_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/norm":0.017248345095268326,"train/train/tensor_act_model_layers_73_self_attn_o_proj/norm":1904.6041544610314,"train/train/tensor_act_model_layers_13_mlp/mean":0.0019817352294921875,"train/train/tensor_grad_model_layers_4_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/max_abs":0.000659942626953125,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/mean":2.0042061805725098e-05,"train/train/tensor_act_model_layers_77_mlp_waleed/std":0.2399906570340138,"train/train/tensor_grad_model_layers_41_mlp_waleed_W_g_weight/max_abs":0.000850677490234375,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_mlp_waleed_W_u_weight/norm":5.8125,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/norm":5,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_38_mlp_waleed_W_g/max_abs":2.21875,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/mean":4.258006811141968e-06,"train/train/layer_model_layers_82/act/norm":24586.25261303083,"train/train/tensor_act_model_layers_77_post_attention_layernorm/max_abs":5.4375,"train/train/tensor_act_model_layers_41_post_attention_layernorm/norm":5792.264282229767,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_37/act/max_abs":25.375,"train/train/tensor_param_model_layers_92_mlp_waleed_W_g_weight/max_abs":0.267578125,"train/train/layer__model_layers_0/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/norm":0.003832417339352984,"train/train/tensor_act_model_layers_61/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_input_layernorm/std":1.000001574021224,"train/train/tensor_act_model_layers_33_self_attn_o_proj/norm":331.91591208265316,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/max_abs":0.2451171875,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_mlp_waleed_W_g_weight/max_abs":0.00040435791015625,"train/train/tensor_act_model_layers_87_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn/norm":760.2183491231991,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/max_abs":0.1826171875,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/max_abs":0.28515625,"train/train/tensor_act_model_layers_14_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/std":0.000690460205078125,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/max_abs":0.1689453125,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_mlp_waleed_W_u_weight/max_abs":0.10107421875,"train/train/tensor_act_model_layers_83_mlp/max_abs":2.796875,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/norm":3.390625,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/max_abs":0.15234375,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/std":8.604666285233523e-05,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_self_attn/mean":-0.0018863677978515625,"train/train/tensor_act_model_layers_26_input_layernorm/mean":0.00171661376953125,"train/train/tensor_act_model_layers_2_mlp_waleed/max_abs":7.28125,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/std":0.04443359375,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/norm":0.003467356274026804,"train/train/tensor_act_model_layers_78_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp_down_proj/norm":330.01048833495656,"train/train/layer__model_layers_3/param/mean":0.001565498792437049,"train/train/tensor_act_model_layers_93_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_v_proj/std":0.6835940423607202,"train/train/layer_model_layers_14/act/norm":20728.49016017801,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_mlp/max_abs":4.34375,"train/train/tensor_act_model_layers_56_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/mean":6.062909960746765e-07,"train/train/tensor_act_model_layers_65_self_attn_o_proj/max_abs":1.1484375,"train/train/tensor_act_model_layers_47_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn/max_abs":1.640625,"train/train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_input_layernorm/std":1.0000015150278507,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/mean":-6.472691893577576e-07,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/norm":4.5625,"train/train/tensor_act_model_layers_10_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_down_proj/max_abs":1.296875,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/max_abs":0.0004482269287109375,"train/train/tensor_act_model_layers_38_input_layernorm/max_abs":5.34375,"train/train/layer_model_layers_27/act/mean":-0.004066258668899536,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/mean":-0.0004177093505859375,"train/train/tensor_act_model_layers_81_self_attn_o_proj/norm":2964.565964577804,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/norm":0.11011527090579996,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_mlp_waleed_W_u_weight/norm":0.013031970665121228,"train/train/tensor_param_model_layers_86_mlp_waleed_W_u_weight/norm":8.3125,"train/train/tensor_act_model_layers_7_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn_q_proj/mean":0.010986328125,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/std":2.7404749615782984e-05,"train/train/tensor_param_model_layers_39_mlp_waleed_W_u_weight/mean":-0.00014019012451171875,"train/train/tensor_act_model_layers_8_mlp_waleed/max_abs":3.078125,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/mean":-1.9747676560655236e-07,"train/train/tensor_param_model_layers_51_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_58_self_attn_k_proj/max_abs":5.375,"train/train/tensor_act_model_layers_41_mlp_waleed_W_g/max_abs":2.390625,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_49/param/mean":0.0014937604645299093,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/std":4.0407370994691986e-05,"train/train/tensor_act_model_layers_71_self_attn/norm":791.4331879325297,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/mean":0.0001850128173828125,"train/train/layer_model_layers_50/act/std":0.8460891016268808,"train/train/tensor_act_model_layers_15_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_k_proj/max_abs":5.09375,"train/train/tensor_act_model_layers_38_mlp_waleed/std":0.0911867724431438,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_waleed_W_g_weight/max_abs":0.0006866455078125,"train/train/tensor_act_model_layers_75_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44/norm":15185.87647571929,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/std":2.077074350101938e-05,"train/train/tensor_act_model_layers_1_self_attn_v_proj/norm":2281.2012348452217,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/std":0.0002033838140891392,"train/train/layer__model_layers_63/param/norm":21.03704182878976,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/std":4.768614981541382e-05,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/mean":-1.178705133497715e-07,"train/train/tensor_act_model_layers_35_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn/norm":1511.1161012052517,"train/train/layer_model_layers_71/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/mean":-0.0004329681396484375,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/max_abs":0.1298828125,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp_waleed/max_abs":3.09375,"train/train/tensor_act_model_layers_17_self_attn_q_proj/norm":8432.621530820512,"train/train/tensor_param_model_layers_9_mlp_waleed_W_u_weight/max_abs":0.10693359375,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/norm":0.041078247221035404,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/mean":-1.198495738208294e-07,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn_k_proj/max_abs":5,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/mean":-7.636845111846924e-07,"train/train/tensor_act_model_layers_27_input_layernorm/max_abs":5.53125,"train/train/tensor_act_model_layers_73_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_mlp_waleed_W_g_weight/std":0.035400390625,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_58/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_mlp_waleed_W_g_weight/max_abs":0.000896453857421875,"train/train/layer_model_layers_28/grad/mean":-3.828992558622881e-09,"train/train/tensor_act_model_layers_22_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_89/act/max_abs":32.5,"train/train/tensor_act_model_layers_73/std":2.789096923673662,"train/train/tensor_act_model_layers_27_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/max_abs":0.216796875,"train/train/tensor_act_model_layers_49_self_attn_v_proj/max_abs":2.015625,"train/train/tensor_param_model_layers_30_mlp_waleed_W_g_weight/norm":4.5,"train/train/tensor_act_model_layers_92_self_attn/mean":-0.0114898681640625,"train/train/tensor_param_model_layers_28_mlp_waleed_W_g_weight/mean":3.647804260253906e-05,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/norm":4.84375,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_waleed_W_g_weight/mean":-0.0003452301025390625,"train/train/layer_model_layers_22/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/max_abs":0.2333984375,"train/train/tensor_param_model_layers_19_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_36_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_mlp_waleed_W_g_weight/max_abs":0.0010986328125,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_27/act/norm":20428.240647658942,"train/train/tensor_grad_model_layers_19_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_waleed_W_g_weight/std":8.28672232195275e-05,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/mean":-7.677078247070312e-05,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/layer_model_layers_80/act/frac_near_user_limit":0,"train/train/layer_model_layers_50/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_mlp_waleed_W_u_weight/max_abs":0.0011444091796875,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/max_abs":0.15625,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/max_abs":9.72747802734375e-05,"train/train/tensor_param_model_layers_0_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer_model_layers_29/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/max_abs":0.333984375,"train/train/tensor_grad_model_layers_78_mlp_waleed_W_g_weight/mean":-4.651956260204315e-07,"train/train/tensor_param_model_layers_79_mlp_waleed_W_u_weight/max_abs":0.2421875,"train/train/tensor_act_model_layers_6_mlp_waleed_W_g/mean":0.00208282470703125,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/max_abs":0.000690460205078125,"train/train/tensor_param_model_layers_73_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_input_layernorm/mean":0.017364501953125,"train/train/tensor_act_model_layers_81_self_attn_o_proj/std":0.5117412452539847,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/std":1.6661612281861576e-05,"train/train/tensor_act_model_layers_37_mlp_waleed/frac_near_dtype_limit":0,"train/train/layer__model_layers_7/param/max_abs":1,"train/train/tensor_param_model_layers_56_mlp_waleed_W_g_weight/std":0.0286865234375,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/mean":9.469687938690186e-06,"train/train/tensor_param_model_layers_15_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_post_attention_layernorm/mean":0.0012085437774658203,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_v_proj/max_abs":2.125,"train/train/tensor_act_model_layers_77_mlp_waleed_W_u/max_abs":3.40625,"train/train/tensor_act_model_layers_9_mlp_waleed_W_u/max_abs":2.578125,"train/train/layer_model_layers_92/act/std":1.583954040338401,"train/train/layer_model_layers_7/grad/max_abs":0.0023956298828125,"train/train/tensor_act_model_layers_52_mlp/norm":470.9768870425397,"train/train/tensor_act_model_layers_39_mlp_waleed_W_u/norm":2422.32229701868,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/std":0.059814453125,"train/train/tensor_param_model_layers_25_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_84/act/norm":28191.252967501365,"train/train/tensor_grad_model_layers_27_mlp_waleed_W_g_weight/norm":0.01457465983929544,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_q_proj/norm":5692.700205587676,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/std":0.00024691675355559766,"train/train/tensor_act_model_layers_75_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_28_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_mlp/norm":535.5182471211212,"train/train/tensor_grad_model_layers_47_mlp_waleed_W_g_weight/mean":1.0561780072748661e-07,"train/train/tensor_param_model_layers_73_mlp_waleed_W_g_weight/mean":-0.00012874603271484375,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/mean":-0.00042724609375,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/norm":0.010434468277328355,"train/train/tensor_act_model_layers_71_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_68_mlp_waleed_W_u_weight/max_abs":0.00084686279296875,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/max_abs":0.00013828277587890625,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_12_self_attn_v_proj/max_abs":2.625,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/max_abs":0.1259765625,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_q_proj/std":1.0410211457397762,"train/train/tensor_grad_model_layers_36_mlp_waleed_W_u_weight/norm":0.013092218855781488,"train/train/layer__model_layers_33/param/std":0.047922695065943224,"train/train/layer__model_layers_88/param/norm":25.485941283872567,"train/train/layer_model_layers_39/act/std":0.8499869557845863,"train/train/tensor_param_model_layers_67_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/max_abs":0.0004100799560546875,"train/train/tensor_param_model_layers_92_mlp_waleed_W_u_weight/norm":10.9375,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/mean":2.014636993408203e-05,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/mean":0.00014495849609375,"train/train/tensor_act_model_layers_11_self_attn_k_proj/norm":6812.408159393417,"train/train/tensor_param_model_layers_28_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/std":0.0322265625,"train/train/time_per_step_avg":1.6282022935524583,"train/train/tensor_act_model_layers_84_input_layernorm/mean":0.0108795166015625,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/std":0.025390625,"train/train/tensor_act_model_layers_70_mlp_down_proj/mean":-0.00115203857421875,"train/train/tensor_grad_model_layers_42_mlp_waleed_W_g_weight/std":3.69012025035061e-05,"train/train/layer_model_layers_23/grad/std":9.26450274818508e-05,"train/train/layer_model_layers_17/act/norm":21819.46250921731,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/std":6.1031913155644e-05,"train/train/layer__model_layers_78/param/max_abs":1,"train/train/tensor_act_model_layers_29_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66/norm":15366.93025087574,"train/train/tensor_act_model_layers_45_self_attn_q_proj/norm":5843.458619027975,"train/train/layer__model_layers_61/param/mean":0.0016491394519062012,"train/train/layer_model_layers_10/act/max_abs":26.25,"train/train/tensor_act_model_layers_11_self_attn_v_proj/max_abs":2.28125,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/max_abs":0.000255584716796875,"train/train/tensor_act_model_layers_41_self_attn_o_proj/max_abs":0.57421875,"train/train/tensor_act_model_layers_93_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_input_layernorm/norm":5792.6131591807825,"train/train/tensor_act_model_layers_20_post_attention_layernorm/std":1.000000017462298,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp_waleed_W_u/norm":3587.819931732047,"train/train/tensor_act_model_layers_11_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/mean":2.3079337552189827e-08,"train/train/tensor_act_model_layers_22_self_attn_q_proj/std":1.1093750065061407,"train/train/tensor_act_model_layers_10_mlp_waleed/max_abs":6.1875,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/norm":0.037854001938008856,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/norm":0.03701408600093674,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/std":0.055908203125,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/norm":0.002048886650129594,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57/norm":15015.188306717173,"train/train/tensor_param_model_layers_45_mlp_waleed_W_u_weight/max_abs":0.138671875,"train/train/tensor_act_model_layers_70_self_attn/norm":1324.7645765981647,"train/train/tensor_act_model_layers_48_mlp_waleed/norm":1115.529108498054,"train/train/tensor_act_model_layers_56_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_input_layernorm/max_abs":5.21875,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/std":1.0000000603904464,"train/train/tensor_act_model_layers_88_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_input_layernorm/max_abs":5.21875,"train/train/tensor_act_model_layers_43_mlp/std":0.06689543684290744,"train/train/tensor_act_model_layers_93_mlp_waleed/std":1.484386307271756,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/mean":0.0002803802490234375,"train/train/tensor_grad_model_layers_11_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_4/grad/frac_near_dtype_limit":0,"train/train/layer__model_layers_15/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/max_abs":0.056884765625,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/norm":0.0012172889648206543,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/max_abs":0.201171875,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/mean":-0.00017452239990234375,"train/train/tensor_act_model_layers_48_mlp_waleed/std":0.13623309923353266,"train/train/tensor_param_model_layers_69_mlp_waleed_W_g_weight/max_abs":0.158203125,"train/train/tensor_act_model_layers_24_mlp_waleed/norm":859.6006304020028,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_88/grad/mean":8.237337498880585e-08,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_18_post_attention_layernorm/std":1.000000012194505,"train/train/tensor_act_model_layers_72_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_56/act/std":0.8317068857203782,"train/train/tensor_grad_model_layers_90_mlp_waleed_W_u_weight/std":0.00010137113614796344,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/mean":-2.9103830456733704e-09,"train/train/layer__model_layers_39/param/norm":20.306976412445675,"train/train/tensor_grad_model_layers_17_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_q_proj/std":1.2304811458114644,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/norm":8.6875,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/std":0.0380859375,"train/train/tensor_act_model_layers_1_mlp/std":2.2852197720917644,"train/train/tensor_act_model_layers_72_mlp_waleed/frac_near_user_limit":0,"train/train/layer__model_layers_89/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/max_abs":0.000606536865234375,"train/train/tensor_act_model_layers_71_mlp_waleed/max_abs":5.65625,"train/train/tensor_param_model_layers_66_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_12_self_attn_o_proj/std":0.0977816773878899,"train/train/tensor_act_model_layers_75_self_attn/max_abs":2.59375,"train/train/tensor_act_model_layers_37_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn/std":0.10107496798048737,"train/train/tensor_act_model_layers_72_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_v_proj/mean":0.005279541015625,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp/max_abs":1.484375,"train/train/tensor_act_model_layers_77_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_mlp_waleed_W_g/norm":3310.881026835341,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/std":3.46423692885386e-05,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/max_abs":0.000102996826171875,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/mean":-0.0002231597900390625,"train/train/tensor_grad_model_layers_63_mlp_waleed_W_u_weight/std":5.525602844425591e-05,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/std":0.0263671875,"train/train/tensor_act_model_layers_13_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/mean":1.8335413187742233e-08,"train/train/tensor_act_model_layers_63_mlp_waleed_W_g/mean":-0.009307861328125,"train/train/tensor_act_model_layers_56_input_layernorm/max_abs":5.03125,"train/train/tensor_param_model_layers_78_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/std":0.022705078125,"train/train/tensor_act_model_layers_68_input_layernorm/max_abs":5.125,"train/train/layer__model_layers_29/param/std":0.04843653635973221,"train/train/layer_model_layers_34/act/norm":19820.20690020725,"train/train/tensor_act_model_layers_5_mlp_waleed_W_u/norm":2549.6206259770233,"train/train/tensor_act_model_layers_52_self_attn_o_proj/std":0.0559139736434932,"train/train/layer__model_layers_81/param/norm":23.669435724157008,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/mean":-3.1562522053718567e-06,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/std":0.0242919921875,"train/train/tensor_act_model_layers_39_mlp/mean":0.00243377685546875,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/norm":6.90625,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/std":3.8394925984476676e-05,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/norm":0.0016570809575160677,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn/std":0.22241252775210488,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_input_layernorm/std":1.0000000131258275,"train/train/layer__model_layers_26/param/norm":19.559080172326613,"train/train/tensor_act_model_layers_34_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/mean":4.076957702636719e-05,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/norm":5.40625,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/max_abs":0.1474609375,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/mean":2.3096799850463867e-07,"train/train/tensor_param_model_layers_66_mlp_waleed_W_g_weight/max_abs":0.177734375,"train/train/tensor_act_model_layers_62_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/norm":0.01594198089139705,"train/train/tensor_act_model_layers_15_self_attn_q_proj/std":1.1171875065843855,"train/train/tensor_act_model_layers_90_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/std":0.02392578125,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_38/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_waleed_W_u/std":0.27148440681236924,"train/train/tensor_grad_model_layers_42_mlp_waleed_W_g_weight/norm":0.013338241592640522,"train/train/layer__model_layers_46/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_mlp/max_abs":0.8828125,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_input_layernorm/max_abs":5.1875,"train/train/layer__model_layers_13/param/max_abs":1,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/max_abs":0.1982421875,"train/train/tensor_grad_model_layers_7_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_mlp_waleed_W_u_weight/norm":8.6875,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_waleed_W_u/mean":0.0020904541015625,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/max_abs":0.2119140625,"train/train/tensor_act_model_layers_23_self_attn_v_proj/std":0.6054687746109496,"train/train/tensor_act_model_layers_55_mlp_waleed_W_g/mean":0.008636474609375,"train/train/tensor_param_model_layers_0_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_42_mlp/std":0.06231699266489778,"train/train/tensor_act_model_layers_77_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/layer_model_layers_18/grad/norm":0.035804657140404234,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_input_layernorm/mean":0.0013272762298583984,"train/train/layer_model_layers_71/act/std":0.8755885243846967,"train/train/tensor_act_model_layers_65_mlp/frac_near_user_limit":0,"train/train/layer__model_layers_43/param/max_abs":1,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/std":0.022705078125,"train/train/layer_model_layers_92/act/max_abs":35.75,"train/train/tensor_param_model_layers_33_mlp_waleed_W_g_weight/norm":4.46875,"train/train/tensor_act_model_layers_43_input_layernorm/std":1.0000012769806776,"train/train/layer_model_layers_9/act/mean":-0.0036001205444335938,"train/train/tensor_param_model_layers_89_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_18_self_attn_k_proj/mean":0.0149688720703125,"train/train/tensor_act_model_layers_33_input_layernorm/max_abs":5.46875,"train/train/tensor_act_model_layers_33_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/mean":6.67572021484375e-05,"train/train/layer__model_layers_66/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_v_proj/norm":3018.0797810040235,"train/train/layer_model_layers_28/grad/max_abs":0.00110626220703125,"train/train/tensor_act_model_layers_6_self_attn_o_proj/norm":998.7894548358049,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/mean":5.103647708892822e-07,"train/train/layer__model_layers_45/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/norm":0.0038008170327112558,"train/train/layer_model_layers_59/act/norm":19748.332534602974,"train/train/tensor_act_model_layers_8_self_attn_v_proj/mean":0.0020008087158203125,"train/train/tensor_act_model_layers_20_self_attn_o_proj/mean":0.00025272369384765625,"train/train/tensor_grad_model_layers_87_mlp_waleed_W_g_weight/max_abs":0.0009307861328125,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/max_abs":0.00101470947265625,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/norm":2.9375,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_v_proj/std":0.5332067204797828,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/max_abs":0.0001697540283203125,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/std":6.300193262158531e-05,"train/train/tensor_act_model_layers_34_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/norm":4.84375,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn_q_proj/max_abs":4.96875,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/max_abs":0.1435546875,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/norm":0.038862479226642696,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/norm":0.011536007830030698,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_q_proj/std":1.093750245230511,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_mlp_waleed_W_g/norm":2507.890874377108,"train/train/layer_model_layers_11/act/mean":-0.0026877522468566895,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/norm":0.0007033695600297201,"train/train/tensor_act_model_layers_43_self_attn/std":0.11597121699279217,"train/train/tensor_act_model_layers_82_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/norm":7,"train/train/tensor_act_model_layers_33_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/norm":0.017305054499262864,"train/train/tensor_act_model_layers_90_mlp/std":0.7558674169252445,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_waleed_W_g_weight/std":0.02685546875,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_o_proj/norm":2180.5780806122607,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_v_proj/max_abs":1.859375,"train/train/tensor_act_model_layers_62_self_attn_o_proj/std":0.08752472353641778,"train/train/tensor_param_model_layers_40_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_7/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/max_abs":0.000652313232421875,"train/train/tensor_param_model_layers_67_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/max_abs":0.12060546875,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/mean":0.00018596649169921875,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/norm":0.004306602304854715,"train/train/tensor_act_model_layers_22/mean":-0.0066070556640625,"train/train/tensor_act_model_layers_65_mlp_waleed_W_u/std":0.3974622416972805,"train/train/tensor_param_model_layers_37_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_waleed_W_g_weight/std":7.000054499262699e-05,"train/train/tensor_param_model_layers_23_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_93_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/norm":0.02472443702565964,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/norm":0.012465151250734875,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/mean":1.3923272490501404e-07,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_waleed_W_u_weight/norm":5.28125,"train/train/tensor_grad_model_layers_64_mlp_waleed_W_g_weight/std":4.667952085550326e-05,"train/train/tensor_grad_model_layers_1_mlp_waleed_W_g_weight/norm":0.2636754918733412,"train/train/tensor_grad_model_layers_48_mlp_waleed_W_u_weight/max_abs":0.00077056884765625,"train/train/tensor_act_model_layers_13_mlp_waleed/frac_near_dtype_limit":0,"train/train/layer_model_layers_9/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34/mean":0.00579833984375,"train/train/layer__model_layers_79/param/std":0.056908787667407264,"train/train/tensor_act_model_layers_71_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_89/grad/norm":0.07191279987799497,"train/train/tensor_param_model_layers_77_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_self_attn_v_proj/mean":-0.001346588134765625,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/max_abs":0.000728607177734375,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/max_abs":0.000583648681640625,"train/train/tensor_act_model_layers_36_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/std":6.008468958353392e-05,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/max_abs":0.000972747802734375,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/max_abs":0.0026092529296875,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_mlp_waleed_W_u_weight/max_abs":0.2275390625,"train/train/tensor_act_model_layers_88_self_attn_o_proj/std":0.37604446692330556,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/norm":4.53125,"train/train/tensor_act_model_layers_43_mlp_down_proj/std":0.06689543684290744,"train/train/tensor_grad_model_layers_11_mlp_waleed_W_g_weight/std":4.726991118950837e-05,"train/train/layer_model_layers_25/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_waleed_W_g_weight/mean":-2.2328458726406097e-07,"train/train/tensor_act_model_layers_43_mlp/norm":387.8941567164412,"train/train/tensor_act_model_layers_45_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/norm":0.004865281103668555,"train/train/tensor_act_model_layers_34_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_waleed_W_g_weight/std":0.023681640625,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/std":0.029296875,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_5/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/std":0.039306640625,"train/train/tensor_act_model_layers_58_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_mlp_waleed_W_g_weight/max_abs":0.00045013427734375,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_mlp_waleed_W_u_weight/max_abs":0.16015625,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/std":0.0233154296875,"train/train/tensor_param_model_layers_35_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_51_mlp_waleed_W_g_weight/max_abs":0.00052642822265625,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/max_abs":0.0002231597900390625,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/norm":0.01605445063432514,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp/mean":0.0007839202880859375,"train/train/tensor_act_model_layers_31_mlp/max_abs":0.462890625,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_43/act/norm":19330.493199540128,"train/train/tensor_act_model_layers_14_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/layer_model_layers_14/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48/max_abs":24,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/std":5.39133526883137e-05,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/mean":5.838228389620781e-08,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_v_proj/norm":3865.9464496342043,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/std":9.347719757424241e-05,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_83/param/max_abs":1,"train/train/tensor_param_model_layers_46_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/max_abs":0.00077056884765625,"train/train/tensor_act_model_layers_18_mlp_waleed_W_u/norm":2102.6259570903508,"train/train/tensor_act_model_layers_76_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_o_proj/mean":0.01806640625,"train/train/tensor_param_model_layers_89_mlp_waleed_W_g_weight/norm":9.125,"train/train/tensor_act_model_layers_70_self_attn_q_proj/mean":0.109130859375,"train/train/tensor_act_model_layers_70_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer_model_layers_14/act/mean":-0.0006420686841011047,"train/train/tensor_act_model_layers_2_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_40/param/std":0.048274354004363876,"train/train/layer__model_layers_46/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_60/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer__model_layers_26/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_post_attention_layernorm/norm":5792.368530283022,"train/train/layer__model_layers_10/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/mean":-2.6635825634002686e-06,"train/train/layer__model_layers_13/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/std":0.00012290096184535998,"train/train/tensor_act_model_layers_86_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_37/act/mean":0.0009811073541641235,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/std":7.310934444542667e-05,"train/train/tensor_act_model_layers_36_post_attention_layernorm/max_abs":5.5,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/norm":4.4375,"train/train/tensor_param_model_layers_72_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_mlp_down_proj/max_abs":0.9296875,"train/train/tensor_act_model_layers_18_post_attention_layernorm/norm":5792.60888672195,"train/train/tensor_act_model_layers_69_mlp_waleed_W_g/max_abs":3.0625,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/norm":0.012317047269420444,"train/train/layer_model_layers_53/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_mlp_down_proj/std":0.5517605304651074,"train/train/layer_model_layers_90/grad/max_abs":0.00167083740234375,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_42_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_40/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_waleed_W_u_weight/std":0.0001576515290022504,"train/train/tensor_act_model_layers_72_mlp_waleed/std":0.1928717738453427,"train/train/tensor_act_model_layers_42_self_attn_k_proj/std":0.9912124186302732,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/mean":8.277129381895065e-08,"train/train/layer_model_layers_0/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_mlp_waleed_W_g_weight/std":0.0252685546875,"train/train/tensor_act_model_layers_48_mlp/max_abs":1.078125,"train/train/tensor_grad_model_layers_24_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_33_mlp_waleed/norm":638.4364085486347,"train/train/tensor_act_model_layers_51_self_attn/mean":-0.0009546279907226562,"train/train/layer_model_layers_40/grad/mean":4.116344075307087e-08,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/max_abs":0.0007171630859375,"train/train/tensor_act_model_layers_52_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/max_abs":0.12451171875,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_47_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_k_proj/std":1.0761774274125493,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/norm":4.40625,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/norm":0.012077599026675349,"train/train/tensor_act_model_layers_54_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_56/act/frac_near_user_limit":0,"train/train/layer_model_layers_38/grad/std":5.1728748910928034e-05,"train/train/tensor_act_model_layers_22_mlp_waleed_W_u/norm":2391.66959799792,"train/train/tensor_act_model_layers_32_self_attn/max_abs":0.455078125,"train/train/tensor_act_model_layers_16_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_mlp_waleed_W_g_weight/max_abs":0.0004558563232421875,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/mean":-0.00011014938354492188,"train/train/tensor_grad_model_layers_61_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_waleed_W_u_weight/std":0.0308837890625,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_mlp/norm":360.68942183749544,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/mean":1.7709098756313324e-06,"train/train/tensor_act_model_layers_6_self_attn_v_proj/max_abs":2.1875,"train/train/tensor_act_model_layers_45_self_attn_v_proj/max_abs":2.890625,"train/train/tensor_act_model_layers_80_self_attn/mean":0.0094451904296875,"train/train/tensor_act_model_layers_26_self_attn_k_proj/norm":5356.514522328965,"train/train/tensor_act_model_layers_48_mlp_waleed_W_u/mean":-0.0015087127685546875,"train/train/tensor_act_model_layers_65_self_attn_k_proj/std":0.9960937888014543,"train/train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/norm":7.65625,"train/train/tensor_act_model_layers_4_self_attn_q_proj/mean":0.0223388671875,"train/train/tensor_param_model_layers_43_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_post_attention_layernorm/mean":-0.005859375,"train/train/tensor_act_model_layers_6_mlp_waleed/std":0.3139690235833474,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/max_abs":0.000804901123046875,"train/train/tensor_act_model_layers_69_self_attn_k_proj/norm":5592.050617717048,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/std":0.00012084252977343009,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/norm":3.28125,"train/train/tensor_act_model_layers_0_self_attn/norm":3743.1841436604545,"train/train/tensor_act_model_layers_40_self_attn_k_proj/mean":-0.04486083984375,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_82/act/max_abs":27.875,"train/train/tensor_grad_model_layers_62_mlp_waleed_W_g_weight/max_abs":0.000659942626953125,"train/train/layer_model_layers_2/grad/max_abs":0.006072998046875,"train/train/tensor_act_model_layers_47_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_68/grad/max_abs":0.00159454345703125,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/max_abs":0.000850677490234375,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_mlp_waleed_W_g_weight/std":0.058837890625,"train/train/tensor_param_model_layers_14_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_91_mlp_waleed_W_u/norm":7423.063212380917,"train/train/tensor_grad_model_layers_10_mlp_waleed_W_g_weight/norm":0.023203155055171114,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/max_abs":0.1669921875,"train/train/tensor_act_model_layers_20_input_layernorm/mean":-0.00623321533203125,"train/train/layer_model_layers_93/grad/mean":-2.85326144345651e-07,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/norm":0.023930140506891417,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/max_abs":0.126953125,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/norm":0.0035281227414425496,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/mean":7.492781151086092e-08,"train/train/tensor_act_model_layers_79_mlp_waleed_W_g/mean":0.00730133056640625,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/max_abs":0.14453125,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/norm":0.00965301436005499,"train/train/tensor_act_model_layers_53/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_40/grad/std":3.708154085534913e-05,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/norm":2.984375,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/mean":-8.9290551841259e-08,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/mean":-5.480833351612091e-07,"train/train/tensor_act_model_layers_19_self_attn_v_proj/std":0.3281250405125295,"train/train/tensor_param_model_layers_70_mlp_waleed_W_g_weight/std":0.033447265625,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/norm":3.09375,"train/train/tensor_act_model_layers_10_mlp_waleed_W_g/std":0.2695312538872594,"train/train/tensor_act_model_layers_10_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/mean":-8.074857760220766e-08,"train/train/tensor_param_model_layers_91_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/mean":-1.3801036402583122e-07,"train/train/tensor_act_model_layers_39_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/norm":0.00574940561897235,"train/train/tensor_act_model_layers_13_self_attn/std":0.08130071005163605,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/norm":4.5625,"train/train/tensor_act_model_layers_66_post_attention_layernorm/std":1.0000011092918826,"train/train/tensor_param_model_layers_40_mlp_waleed_W_u_weight/max_abs":0.1494140625,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/max_abs":0.00012302398681640625,"train/train/tensor_act_model_layers_54_self_attn_q_proj/mean":0.028839111328125,"train/train/tensor_act_model_layers_11_mlp_waleed/norm":715.7056474011803,"train/train/tensor_act_model_layers_70_mlp_waleed_W_u/norm":3537.9595264390414,"train/train/tensor_grad_model_layers_14_mlp_waleed_W_g_weight/mean":1.1129304766654968e-07,"train/train/tensor_act_model_layers_32_mlp_waleed_W_u/norm":2394.9893081594405,"train/train/tensor_act_model_layers_58_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_q_proj/norm":4588.023499963439,"train/train/tensor_act_model_layers_5_mlp_waleed/norm":1455.5542180040404,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/mean":-0.00023746490478515625,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/max_abs":0.0013580322265625,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/std":0.02978515625,"train/train/tensor_act_model_layers_0_input_layernorm/max_abs":7.09375,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/std":4.547881709240462e-05,"train/train/tensor_param_model_layers_79_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_mlp_waleed_W_u_weight/std":4.536308319775367e-05,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/std":0.0322265625,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/std":0.02734375,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn/max_abs":8.1875,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/mean":1.8830178305506706e-08,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/norm":0.0008660539051443119,"train/train/tensor_act_model_layers_25_post_attention_layernorm/max_abs":5.4375,"train/train/tensor_act_model_layers_49_post_attention_layernorm/max_abs":5.25,"train/train/tensor_act_model_layers_86_self_attn_o_proj/norm":2046.6089941697269,"train/train/tensor_act_model_layers_34/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/norm":3.140625,"train/train/layer_model_layers_36/grad/max_abs":0.00115966796875,"train/train/tensor_act_model_layers_53_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/std":2.9935941154956825e-05,"train/train/tensor_act_model_layers_12_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_q_proj/std":0.7578175613084368,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_mlp_waleed_W_g_weight/max_abs":0.0011444091796875,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_25/grad/std":4.4132814713682826e-05,"train/train/tensor_act_model_layers_73_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp_waleed_W_g/norm":2656.798560724644,"train/train/tensor_act_model_layers_88_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/norm":0.013932870395322362,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_post_attention_layernorm/norm":5792.613647461692,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_55/max_abs":23.75,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/norm":0.013046791110654207,"train/train/tensor_act_model_layers_19_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_43_mlp_waleed_W_u/mean":-0.0074005126953125,"train/train/tensor_act_model_layers_14_self_attn_o_proj/max_abs":0.88671875,"train/train/tensor_act_model_layers_46/norm":15201.754144474917,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/norm":3.0625,"train/train/tensor_act_model_layers_9_input_layernorm/max_abs":5,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_55/param/std":0.05022041233282669,"train/train/tensor_act_model_layers_88_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_v_proj/std":0.3886719295127869,"train/train/layer_model_layers_74/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_k_proj/max_abs":5,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/mean":5.943002179265022e-08,"train/train/tensor_act_model_layers_41_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp_waleed_W_u/max_abs":3.046875,"train/train/layer_model_layers_36/grad/norm":0.04515431271023093,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/max_abs":0.150390625,"train/train/tensor_act_model_layers_44_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_rotary_emb/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/mean":2.2543827071785927e-07,"train/train/tensor_act_model_layers_85_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/norm":0.012155155581263588,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/mean":-0.0001735687255859375,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/mean":0.00011968612670898438,"train/train/tensor_act_model_layers_37_input_layernorm/mean":0.002414703369140625,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/mean":5.278736352920532e-06,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/mean":3.62396240234375e-05,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/max_abs":0.267578125,"train/train/layer_model_layers_48/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_down_proj/std":0.07959060596245711,"train/train/tensor_act_model_layers_86_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_input_layernorm/max_abs":5.09375,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_post_attention_layernorm/std":1.0000011110290155,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/max_abs":0.001739501953125,"train/train/tensor_param_model_layers_4_mlp_waleed_W_u_weight/norm":4.25,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/std":0.04296875,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/max_abs":0.171875,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp_waleed_W_u/mean":0.003467559814453125,"train/train/tensor_act_model_layers_17_self_attn_v_proj/max_abs":3.09375,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/std":0.042236328125,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_79_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_waleed_W_u_weight/norm":0.026646092656424352,"train/train/tensor_param_model_layers_35_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/max_abs":0.00012683868408203125,"train/train/tensor_act_model_layers_35/mean":0.00836944580078125,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/mean":0.00011396408081054688,"train/train/layer__model_layers_32/param/mean":0.0016500886628482718,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/norm":3.109375,"train/train/tensor_act_model_layers_75/std":2.851596114004602,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_waleed_W_u_weight/max_abs":0.000934600830078125,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/max_abs":0.1474609375,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/norm":5.96875,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_15_self_attn_k_proj/mean":0.00356292724609375,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/std":0.05419921875,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/norm":0.004863103189234402,"train/train/tensor_act_model_layers_55_mlp_down_proj/norm":496.48867507303925,"train/train/tensor_param_model_layers_5_mlp_waleed_W_u_weight/max_abs":0.12353515625,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/std":2.277219311376162e-05,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_norm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/std":9.945860731269019e-05,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/std":2.1786463150398304e-05,"train/train/tensor_param_model_layers_80_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn_o_proj/norm":1628.3842667660665,"train/train/tensor_grad_model_layers_84_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn/mean":-0.00035858154296875,"train/train/tensor_param_model_layers_32_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_49/grad/std":3.778245155707873e-05,"train/train/layer_model_layers_77/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/norm":3.15625,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/mean":5.2675604820251465e-06,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/mean":0.0003681182861328125,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/mean":7.2479248046875e-05,"train/train/tensor_act_model_layers_29_self_attn_o_proj/norm":773.8830190003374,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_43_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_post_attention_layernorm/max_abs":4.90625,"train/train/tensor_grad_model_layers_85_mlp_waleed_W_u_weight/max_abs":0.000804901123046875,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/norm":5.8125,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_42_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/mean":0.000247955322265625,"train/train/tensor_act_model_layers_38_mlp_waleed_W_g/std":0.2851562657878309,"train/train/layer__model_layers_50/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_input_layernorm/mean":-0.011138916015625,"train/train/tensor_act_model_layers_71_self_attn_v_proj/mean":0.00702667236328125,"train/train/tensor_act_model_layers_1/std":3.433636109340838,"train/train/tensor_act_model_layers_59/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/max_abs":0.0003185272216796875,"train/train/tensor_grad_model_layers_35_mlp_waleed_W_g_weight/max_abs":0.00051116943359375,"train/train/tensor_act_model_layers_24_self_attn_q_proj/std":1.0390625173660148,"train/train/tensor_act_model_layers_23_self_attn/max_abs":2.75,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/std":5.454414921305537e-05,"train/train/tensor_act_model_layers_50_input_layernorm/mean":-0.00030994415283203125,"train/train/tensor_param_model_layers_36_mlp_waleed_W_u_weight/mean":1.3530254364013672e-05,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/max_abs":6.771087646484375e-05,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_q_proj/mean":-0.016143798828125,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/max_abs":0.00135040283203125,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp/mean":-0.00018227100372314453,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/norm":0.014655874271904295,"train/train/tensor_param_model_layers_30_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_k_proj/max_abs":5.40625,"train/train/tensor_param_model_layers_51_mlp_waleed_W_g_weight/mean":-0.00020599365234375,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp_down_proj/max_abs":0.462890625,"train/train/tensor_act_model_layers_30_mlp_down_proj/max_abs":0.474609375,"train/train/layer_model_layers_5/grad/std":0.00010675653302358734,"train/train/tensor_act_model_layers_57_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/mean":0.000644683837890625,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn/mean":0.00330352783203125,"train/train/tensor_act_model_layers_90_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_input_layernorm/max_abs":5.875,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp_waleed/norm":2842.2043399913,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/max_abs":0.12890625,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_mlp_waleed_W_g_weight/mean":-1.048319973051548e-07,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/std":4.470493871588318e-05,"train/train/tensor_act_model_layers_77_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_v_proj/max_abs":3.15625,"train/train/tensor_act_model_layers_68_self_attn_v_proj/std":0.3818374375218736,"train/train/layer_model_layers_46/act/max_abs":24.5,"train/train/tensor_act_model_layers_90_post_attention_layernorm/std":1.000000214049555,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/max_abs":0.1328125,"train/train/layer_model_layers_46/act/std":0.8307432413716219,"train/train/tensor_param_model_layers_59_input_layernorm_weight/mean":1,"train/train/layer__model_layers_23/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/max_abs":0.0006103515625,"train/train/tensor_act_model_layers_64_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_waleed_W_u/std":0.2558593913799019,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/norm":0.014570825327421145,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_63_input_layernorm/std":1.0000011254444903,"train/train/tensor_act_model_layers_1_mlp_waleed/norm":8159.832790059655,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_65_self_attn_q_proj/mean":0.0714111328125,"train/train/tensor_param_model_layers_54_mlp_waleed_W_u_weight/norm":5.03125,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/norm":4.75,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/max_abs":0.000335693359375,"train/train/tensor_act_model_layers_92/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp_waleed_W_u/norm":2303.822414150166,"train/train/tensor_param_model_layers_82_mlp_waleed_W_g_weight/norm":7.40625,"train/train/tensor_grad_model_layers_82_mlp_waleed_W_g_weight/std":7.748940194266859e-05,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_mlp_waleed_W_g_weight/norm":4.1875,"train/train/tensor_act_model_layers_65_mlp_down_proj/std":0.1069336280430778,"train/train/tensor_act_model_layers_43_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp/max_abs":0.8515625,"train/train/tensor_param_model_layers_77_mlp_waleed_W_u_weight/mean":0.00014972686767578125,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/mean":3.0817463994026184e-06,"train/train/tensor_param_model_layers_62_mlp_waleed_W_u_weight/norm":5.5,"train/train/layer_model_layers_16/act/std":0.9216824483074956,"train/train/tensor_act_model_layers_40_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_mlp_waleed_W_g_weight/norm":4.1875,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/std":0.04150390625,"train/train/tensor_act_model_layers_61_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_q_proj/norm":7168.411637924817,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/norm":4.59375,"train/train/tensor_param_model_layers_61_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_waleed_W_u_weight/max_abs":0.236328125,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/norm":0.02404955552933567,"train/train/layer_model_layers_21/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_waleed_W_g/mean":-0.00212860107421875,"train/train/tensor_act_model_layers_41_mlp_waleed_W_u/max_abs":2.078125,"train/train/layer__model_layers_37/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_v_proj/mean":0.00556182861328125,"train/train/tensor_act_model_layers_46_self_attn_v_proj/max_abs":2.578125,"train/train/layer_model_layers_20/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_mlp_waleed_W_u_weight/max_abs":0.27734375,"train/train/layer_model_layers_18/grad/std":4.4237356048571496e-05,"train/train/tensor_grad_model_layers_90_mlp_waleed_W_u_weight/max_abs":0.0008697509765625,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/max_abs":0.00093841552734375,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_44/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp_waleed/frac_near_dtype_limit":0,"train/train/layer_model_layers_64/grad/norm":0.049487964268632834,"train/train/tensor_act_model_layers_56_self_attn_k_proj/mean":-0.001987457275390625,"train/train/tensor_act_model_layers_31/mean":0.0040111541748046875,"train/train/tensor_act_model_layers_22/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/max_abs":0.00010204315185546875,"train/train/layer__model_layers_91/param/norm":27.270688104400666,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/std":8.620648360870185e-05,"train/train/tensor_act_model_layers_92_self_attn/norm":3668.9699777493547,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/mean":0.000156402587890625,"train/train/tensor_act_model_layers_3_mlp_waleed/mean":0.0082550048828125,"train/train/tensor_param_model_layers_74_mlp_waleed_W_g_weight/norm":6.40625,"train/train/layer_model_layers_75/grad/max_abs":0.00148773193359375,"train/train/tensor_act_model_layers_13_self_attn_k_proj/norm":7069.955824252163,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_waleed_W_u/mean":-0.0026416778564453125,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/std":5.9009739820180714e-05,"train/train/layer_model_layers_31/grad/mean":-8.076390138189431e-08,"train/train/layer__model_layers_14/param/frac_near_user_limit":0,"train/train/layer_model_layers_53/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp/std":0.13623138613970398,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/mean":5.539506673812866e-06,"train/train/layer_model_layers_47/act/std":0.8159733505561675,"train/train/tensor_act_model_layers_1_self_attn_k_proj/max_abs":5.125,"train/train/tensor_param_model_layers_78_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer__model_layers_56/param/mean":0.0015116775649572126,"train/train/tensor_act_model_layers_80_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/norm":3.203125,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/mean":1.0235235095024109e-06,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp_waleed_W_u/std":0.38379039836971374,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/max_abs":0.00019168853759765625,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/mean":6.239861249923706e-08,"train/train/tensor_act_model_layers_67_input_layernorm/mean":0.006805419921875,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/std":4.466027240604462e-05,"train/train/tensor_act_model_layers_66/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_waleed/norm":1366.061302400292,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/mean":6.580352783203125e-05,"train/train/layer_model_layers_73/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_k_proj/norm":5490.337444110409,"train/train/tensor_act_model_layers_11_self_attn/mean":-0.0002193450927734375,"train/train/tensor_act_model_layers_62_mlp/std":0.09460481517705308,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/max_abs":0.00160980224609375,"train/train/tensor_act_model_layers_54_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_post_attention_layernorm/norm":5792.613769537218,"train/train/tensor_act_model_layers_85_mlp_down_proj/std":0.4179757175577836,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_q_proj/max_abs":5.0625,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/std":5.359619809744314e-05,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_mlp_waleed_W_u_weight/mean":-1.5139812603592873e-07,"train/train/tensor_act_model_layers_75_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/std":1.142993256851609e-05,"train/train/tensor_param_model_layers_22_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_mlp_down_proj/mean":0.00017058849334716797,"train/train/tensor_act_model_layers_38_mlp_waleed_W_u/max_abs":2.515625,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/mean":-0.00010919570922851562,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/mean":6.362795829772949e-06,"train/train/tensor_act_model_layers_55_self_attn_k_proj/mean":-0.02703857421875,"train/train/tensor_act_model_layers_29_self_attn/max_abs":1.1015625,"train/train/tensor_param_model_layers_21_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_7/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/mean":0.0004825592041015625,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_input_layernorm/norm":5792.611572267163,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_52/std":2.6133156346904824,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/max_abs":0.00124359130859375,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/std":0.0245361328125,"train/train/tensor_act_model_layers_74_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/max_abs":0.11474609375,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/norm":6.09375,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/max_abs":0.279296875,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/std":0.00010400358409762954,"train/train/layer_model_layers_2/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/std":0.04443359375,"train/train/tensor_param_model_layers_62_mlp_waleed_W_u_weight/std":0.0302734375,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/norm":5.21875,"train/train/tensor_act_model_layers_84_self_attn_v_proj/mean":-0.00626373291015625,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp_waleed/mean":0.00122833251953125,"train/train/layer_model_layers_20/grad/max_abs":0.0011444091796875,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/max_abs":0.11279296875,"train/train/tensor_act_model_layers_25_post_attention_layernorm/mean":0.0017757415771484375,"train/train/tensor_act_model_layers_36_self_attn_q_proj/std":1.07812502073205,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_v_proj/std":0.30957188440399125,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/std":0.028564453125,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_q_proj/max_abs":5.75,"train/train/tensor_act_model_layers_32_self_attn/std":0.024964390894092397,"train/train/tensor_act_model_layers_69_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/std":1.3544380185008781e-05,"train/train/tensor_act_model_layers_17_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/norm":0.010546859105416083,"train/train/tensor_act_model_layers_83_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_q_proj/mean":0.003681182861328125,"train/train/tensor_act_model_layers_64_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/mean":0.00014781951904296875,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_25_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/max_abs":0.0015411376953125,"train/train/tensor_param_model_layers_89_mlp_waleed_W_g_weight/max_abs":0.248046875,"train/train/tensor_act_model_layers_51_mlp_waleed/norm":1050.3747727829093,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/max_abs":0.2216796875,"train/train/tensor_act_model_layers_44_mlp/max_abs":0.59375,"train/train/tensor_act_model_layers_71_mlp_waleed_W_g/std":0.43798942171063143,"train/train/tensor_act_model_layers_20_mlp_waleed_W_u/mean":-0.003200531005859375,"train/train/tensor_act_model_layers_14_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/layer__model_layers_34/param/max_abs":1,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/mean":7.203198038041592e-09,"train/train/tensor_act_model_layers_42/max_abs":24.75,"train/train/tensor_act_model_layers_92_self_attn/max_abs":6.53125,"train/train/tensor_act_model_layers_27_mlp_waleed_W_u/norm":2221.85644304563,"train/train/layer_model_layers_42/act/norm":19548.838240487094,"train/train/layer_model_layers_93/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_87/act/std":1.178443648753365,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/norm":5.0625,"train/train/layer_model_layers_27/act/max_abs":26.5,"train/train/tensor_act_model_layers_62_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/max_abs":0.00014591217041015625,"train/train/tensor_act_model_layers_77_mlp_waleed_W_g/max_abs":3.09375,"train/train/tensor_param_model_layers_29_mlp_waleed_W_g_weight/mean":-0.00010967254638671875,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/std":3.8434467605612486e-05,"train/train/tensor_act_model_layers_80_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_63_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_waleed_W_u_weight/norm":0.0245990610459091,"train/train/tensor_param_model_layers_91_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_waleed_W_g/max_abs":2.921875,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/norm":0.015106725202006096,"train/train/tensor_param_model_layers_9_mlp_waleed_W_u_weight/std":0.0230712890625,"train/train/tensor_param_model_layers_79_input_layernorm_weight/max_abs":1,"train/train/layer__model_layers_16/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_mlp_waleed/std":0.4365267913151347,"train/train/tensor_act_model_layers_10_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_mlp_waleed_W_u_weight/norm":5.65625,"train/train/tensor_act_model_layers_91_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12/std":2.980497036488133,"train/train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/std":0.054443359375,"train/train/tensor_act_model_layers_33/mean":0.0039234161376953125,"train/train/tensor_param_model_layers_79_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/std":2.491546099735534e-05,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/std":1.9221741315009918e-05,"train/train/tensor_grad_model_layers_77_mlp_waleed_W_u_weight/max_abs":0.00058746337890625,"train/train/tensor_act_model_layers_10_mlp_down_proj/std":0.11169509005992263,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/max_abs":0.0002841949462890625,"train/train/tensor_act_model_layers_82_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn_q_proj/max_abs":6.875,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/mean":0.00026702880859375,"train/train/tensor_act_model_layers_8_self_attn_o_proj/mean":-0.0010700225830078125,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/std":0.034912109375,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/std":5.2294606597957e-05,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp/max_abs":0.458984375,"train/train/layer__model_layers_22/param/mean":0.0014929868129783786,"train/train/tensor_act_model_layers_38_self_attn_k_proj/std":1.0390625766345403,"train/train/tensor_act_model_layers_25_self_attn_v_proj/mean":-0.001430511474609375,"train/train/tensor_act_model_layers_85_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/norm":2.875,"train/train/tensor_act_model_layers_34_self_attn_k_proj/max_abs":4.90625,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/std":3.811796544864576e-05,"train/train/layer__model_layers_3/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/norm":0.0221127599814115,"train/train/tensor_param_model_layers_52_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_waleed_W_u/std":0.2675781667667551,"train/train/tensor_act_model_layers_84_self_attn_o_proj/max_abs":7.53125,"train/train/tensor_act_model_layers_5_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_v_proj/norm":1792.5300149294421,"train/train/tensor_act_model_layers_4_mlp_waleed/mean":0.0027313232421875,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/norm":0.031257509283012394,"train/train/tensor_act_model_layers_2_self_attn/std":0.07153380860289689,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/std":2.5382968311988594e-05,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/norm":0.0008173139118963859,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/mean":1.253560185432434e-06,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/max_abs":0.25390625,"train/train/tensor_param_model_layers_75_mlp_waleed_W_u_weight/std":0.036376953125,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_77_mlp_waleed_W_u/std":0.4980468852847229,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/std":7.660532173857849e-05,"train/train/tensor_act_model_layers_84_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_norm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_47/act/max_abs":24.5,"train/train/tensor_act_model_layers_66_self_attn_o_proj/std":0.22583156804289212,"train/train/tensor_param_model_layers_34_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/std":0.00010276233556834098,"train/train/tensor_grad_model_layers_16_mlp_waleed_W_u_weight/mean":9.726500138640404e-08,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/norm":6.5625,"train/train/tensor_grad_model_layers_85_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp_waleed_W_u/std":0.5751979700512888,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/std":0.037841796875,"train/train/tensor_act_model_layers_60_self_attn_q_proj/mean":-0.02886962890625,"train/train/tensor_act_model_layers_35_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/norm":3.046875,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/mean":3.416789695620537e-08,"train/train/tensor_grad_model_layers_40_mlp_waleed_W_u_weight/max_abs":0.00106048583984375,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/max_abs":0.2177734375,"train/train/tensor_act_model_layers_48_mlp_waleed_W_g/max_abs":3.3125,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/max_abs":0.000850677490234375,"train/train/tensor_act_model_layers_73_self_attn_o_proj/max_abs":2.78125,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/mean":9.516952559351921e-08,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/std":9.108725261688183e-05,"train/train/tensor_param_model_layers_27_mlp_waleed_W_u_weight/max_abs":0.130859375,"train/train/tensor_grad_model_layers_92_mlp_waleed_W_u_weight/norm":0.045362385880176145,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_waleed_W_u/mean":0.0010933876037597656,"train/train/layer_model_layers_91/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp/norm":352.85763489002466,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_mlp_waleed_W_g_weight/norm":0.016058162860661668,"train/train/tensor_act_model_layers_87_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_o_proj/norm":2511.450679147774,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/norm":6.46875,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/max_abs":0.00020122528076171875,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp/max_abs":4.8125,"train/train/tensor_grad_model_layers_83_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_mlp_waleed_W_g_weight/max_abs":0.000946044921875,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/mean":0.000179290771484375,"train/train/tensor_act_model_layers_55_self_attn_k_proj/std":0.8574242407173072,"train/train/tensor_act_model_layers_83_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_waleed_W_u/max_abs":3.234375,"train/train/layer__model_layers_84/param/mean":0.0016988376373433844,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/std":6.145576093437144e-05,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/max_abs":0.2373046875,"train/train/tensor_act_model_layers_26_self_attn_k_proj/mean":-0.03375244140625,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/max_abs":0.00139617919921875,"train/train/tensor_act_model_layers_85_self_attn/max_abs":4.125,"train/train/tensor_act_model_layers_38_self_attn_v_proj/max_abs":3.5625,"train/train/tensor_grad_model_layers_24_mlp_waleed_W_g_weight/std":4.3463029640595026e-05,"train/train/tensor_act_model_layers_84_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp/std":0.08215359973647395,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/norm":2.96875,"train/train/tensor_grad_model_layers_74_mlp_waleed_W_g_weight/norm":0.02249184844916559,"train/train/tensor_act_model_layers_71_self_attn_o_proj/mean":0.0009326934814453125,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/std":1.8651412260902216e-05,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/mean":0.0002689361572265625,"train/train/tensor_param_model_layers_93_mlp_waleed_W_u_weight/max_abs":0.318359375,"train/train/tensor_act_model_layers_69_self_attn_k_proj/mean":0.067138671875,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/max_abs":0.0003528594970703125,"train/train/tensor_grad_model_layers_5_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_mlp_waleed_W_g_weight/std":0.0234375,"train/train/tensor_act_model_layers_12/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/norm":0.017273057054378496,"train/train/tensor_act_model_layers_40_self_attn_k_proj/norm":4588.361407923702,"train/train/tensor_grad_model_layers_44_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/std":0.024658203125,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/norm":0.0013534651264524554,"train/train/tensor_grad_model_layers_50_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn_q_proj/max_abs":5.4375,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/std":5.4006849048923905e-05,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_post_attention_layernorm/norm":5792.609619142893,"train/train/tensor_act_model_layers_61_mlp_waleed_W_g/norm":3154.2834372658367,"train/train/tensor_act_model_layers_9_mlp_down_proj/std":0.09399446039950485,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/std":4.6330423657446945e-05,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_v_proj/mean":-0.0044708251953125,"train/train/tensor_grad_model_layers_23_mlp_waleed_W_u_weight/max_abs":0.000766754150390625,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/mean":1.7762184143066406e-05,"train/train/tensor_act_model_layers_79_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_57/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp/norm":455.54596648817636,"train/train/tensor_param_model_layers_24_mlp_waleed_W_g_weight/mean":-0.0001926422119140625,"train/train/tensor_param_model_layers_65_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_93/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_72/grad/std":7.765572114882264e-05,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_embed_tokens_weight/std":0.078125,"train/train/tensor_act_model_layers_69_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_10/grad/max_abs":0.0030517578125,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_mlp_waleed_W_u_weight/max_abs":0.1494140625,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/max_abs":0.000690460205078125,"train/train/layer_model_layers_36/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/norm":3.421875,"train/train/tensor_grad_model_layers_19_mlp_waleed_W_u_weight/mean":-7.275957614183426e-08,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/norm":4.9375,"train/train/tensor_act_model/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_k_proj/std":0.8476564669938403,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/norm":7.6875,"train/train/layer__model_layers_62/param/frac_near_user_limit":0,"train/train/layer__model_layers_74/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_input_layernorm/mean":0.0034122467041015625,"train/train/tensor_param_model_layers_31_mlp_waleed_W_g_weight/max_abs":0.126953125,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/norm":3.015625,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/norm":4.875,"train/train/tensor_act_model_layers_25_mlp_waleed/std":0.09387231653981724,"train/train/layer_model_layers_17/grad/max_abs":0.0026397705078125,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/norm":0.021888997504149865,"train/train/tensor_act_model_layers_55_post_attention_layernorm/max_abs":5.03125,"train/train/tensor_act_model_layers_79_mlp_down_proj/std":0.20898439293883872,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp_waleed/max_abs":6.4375,"train/train/tensor_param_model_layers_59_mlp_waleed_W_g_weight/mean":1.4528632164001465e-06,"train/train/tensor_act_model_layers_6_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/max_abs":0.00110626220703125,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/norm":6.5625,"train/train/layer__model_layers_28/param/mean":0.0015017885871684868,"train/train/tensor_param_model_layers_54_mlp_waleed_W_u_weight/std":0.0277099609375,"train/train/tensor_grad_model_layers_44_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_mlp_waleed_W_g_weight/max_abs":0.000762939453125,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/mean":-1.2524425983428955e-05,"train/train/layer_model_layers_64/act/mean":0.005669713020324707,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/std":2.4338750180634602e-05,"train/train/tensor_act_model_layers_84/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_mlp_waleed_W_g_weight/norm":0.04706360701921905,"train/train/layer_model_layers_25/grad/mean":1.094774408454828e-07,"train/train/tensor_grad_model_layers_52_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_32/param/std":0.04750980247772962,"train/train/tensor_act_model_layers_69_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_23_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn/norm":3239.3336442728105,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/mean":1.0251998901367188e-05,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_77_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/norm":5095.901786977963,"train/train/layer_model_layers_90/act/norm":30380.575839999423,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/norm":0.0013759111205431786,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/norm":0.0014172870994890467,"train/train/tensor_act_model_layers_46_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/max_abs":0.000518798828125,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/norm":0.0036730681798987056,"train/train/tensor_grad_model_layers_57_mlp_waleed_W_g_weight/norm":0.015750086295935956,"train/train/tensor_act_model_layers_0_mlp_waleed/norm":11050.327818431557,"train/train/tensor_act_model_layers_31/norm":15548.602289682358,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/norm":0.002447704297941703,"train/train/tensor_act_model_layers_45/norm":15177.916796873704,"train/train/tensor_act_model_layers_51_mlp_waleed/frac_near_user_limit":0,"train/train/layer_model_layers_72/grad/mean":2.4087619553676074e-07,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_v_proj/norm":3317.8583133673674,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/std":0.02685546875,"train/train/tensor_act_model_layers_22_mlp_waleed_W_u/max_abs":2.6875,"train/train/layer_model_layers_47/grad/norm":0.026927039854167573,"train/train/layer_model_layers_30/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_down_proj/std":0.046875327021018556,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/norm":4.34375,"train/train/layer_model_layers_54/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/max_abs":0.0004444122314453125,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/max_abs":0.00019359588623046875,"train/train/tensor_act_model_layers_85_mlp_waleed/mean":-0.00342559814453125,"train/train/tensor_param_model_layers_5_mlp_waleed_W_g_weight/norm":4.1875,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/max_abs":0.12353515625,"train/train/tensor_grad_model_layers_10_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/norm":0.001610570426003565,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/mean":-3.337860107421875e-05,"train/train/tensor_param_model_layers_83_mlp_waleed_W_g_weight/mean":0.0002422332763671875,"train/train/tensor_param_model_layers_14_mlp_waleed_W_u_weight/mean":-0.00016117095947265625,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/norm":4.46875,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/norm":4.9375,"train/train/tensor_act_model_layers_53_mlp_waleed/max_abs":3.234375,"train/train/layer__model_layers_90/param/std":0.06611608569642453,"train/train/tensor_act_model_layers_79_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn_k_proj/norm":7541.460011527253,"train/train/tensor_grad_model_layers_75_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/std":3.982058066362149e-05,"train/train/tensor_act_model_layers_56_self_attn_k_proj/std":0.8330096745054453,"train/train/tensor_act_model_layers_53_mlp_waleed_W_u/max_abs":2.484375,"train/train/tensor_act_model_layers_26_mlp_waleed_W_g/mean":-0.0019512176513671875,"train/train/tensor_act_model_layers_54_post_attention_layernorm/mean":0.0006401538848876953,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/max_abs":0.00096893310546875,"train/grad_norm":1.7265625,"train/train/tensor_param_model_layers_36_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_mlp_waleed_W_u_weight/max_abs":0.1396484375,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/std":0.05078125,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/norm":0.019610815109322055,"train/train/layer_model_layers_26/grad/std":4.635981949574991e-05,"train/train/tensor_act_model_layers_15_self_attn_k_proj/norm":6066.794413071633,"train/train/tensor_act_model_layers_25_self_attn/mean":-0.0014219284057617188,"train/train/tensor_act_model_layers_3_mlp_down_proj/norm":3382.6723212187767,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_k_proj/std":0.9755882127361095,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp/norm":4383.015450001458,"train/train/tensor_act_model_layers_6_post_attention_layernorm/max_abs":4.84375,"train/train/tensor_act_model_layers_15_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/mean":8.689239621162415e-07,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/std":0.04541015625,"train/train/tensor_act_model_layers_84/norm":19992.57041459343,"train/train/tensor_act_model_layers_12/mean":-0.0089569091796875,"train/train/tensor_act_model_layers_69/mean":0.01214599609375,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/max_abs":9.5367431640625e-05,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/std":0.0001160447583942888,"train/train/layer__model_layers_42/param/mean":0.0014904099581952027,"train/train/tensor_act_model_layers_2_self_attn_v_proj/std":0.3334972054664044,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/mean":-1.5343539416790009e-07,"train/train/layer_model_layers_66/grad/frac_near_user_limit":0,"train_loss":14.663889963785808,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_28/param/norm":19.943744124098163,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/max_abs":0.00091552734375,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn/std":0.5400419810718082,"train/train/tensor_grad_model_layers_17_mlp_waleed_W_u_weight/norm":0.017644755917939358,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/std":0.0390625,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/mean":-8.021015673875809e-08,"train/train/tensor_act_model_layers_17_post_attention_layernorm/mean":-0.007293701171875,"train/train/tensor_act_model_layers_48_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_waleed_W_g/std":0.3417969190794946,"train/train/tensor_act_model_layers_11_mlp_waleed/std":0.08728053152939844,"train/train/tensor_act_model_layers_40_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_62/grad/norm":0.03797046092971983,"train/train/tensor_act_model_layers_75_mlp_waleed_W_g/norm":3973.7667648080696,"train/train/global/grad/norm":2.779114401995687,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/norm":0.01399433117222351,"train/train/tensor_act_model_layers_0_self_attn/max_abs":3.15625,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/std":1.9860254061371346e-05,"train/train/tensor_act_model_layers_0_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_post_attention_layernorm/mean":0.0011649131774902344,"train/train/tensor_grad_model_layers_40_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_mlp_waleed_W_g_weight/norm":0.01433134192417707,"train/train/tensor_grad_model_layers_61_mlp_waleed_W_g_weight/std":4.786540423449261e-05,"train/train/tensor_act_model_layers_44_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/mean":-6.915070116519928e-08,"train/train/tensor_grad_model_layers_73_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/norm":0.0035556416488565453,"train/train/layer__model_layers_92/param/std":0.06778162996891919,"train/train/tensor_act_model_layers_73_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/std":0.0262451171875,"train/train/tensor_grad_model_layers_30_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_o_proj/std":0.07153380860289689,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_mlp_waleed_W_u_weight/max_abs":0.2578125,"train/train/tensor_act_model_layers_9_post_attention_layernorm/std":1.0000000533182158,"train/train/tensor_param_model_layers_11_mlp_waleed_W_g_weight/mean":0.00011491775512695312,"train/train/tensor_act_model_layers_69_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_mlp_waleed_W_u_weight/std":7.677826337438137e-05,"train/train/tensor_act_model_layers_65_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/std":3.333022927428295e-05,"train/train/tensor_act_model_layers_38_self_attn/norm":1374.518252251199,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/std":0.024658203125,"train/train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp_waleed_W_u/norm":2981.8872294572584,"train/train/tensor_param_model_layers_55_mlp_waleed_W_u_weight/mean":0.00011348724365234375,"train/train/tensor_param_model_layers_88_mlp_waleed_W_u_weight/std":0.049072265625,"train/train/tensor_act_model_layers_46/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_mlp_waleed_W_g/norm":2668.069765464196,"train/train/layer_model_layers_49/act/mean":0.0032815858721733093,"train/train/layer__model_layers_33/param/max_abs":1,"train/train/tensor_param_model_layers_77_mlp_waleed_W_u_weight/std":0.03759765625,"train/train/tensor_act_model_layers_8/norm":17917.061998118697,"train/train/layer_model_layers_62/grad/max_abs":0.0007781982421875,"train/train/tensor_act_model_layers_72_self_attn_v_proj/std":0.4785156590597958,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/mean":2.4389009922742844e-08,"train/train/tensor_act_model_layers_80_mlp_waleed_W_u/mean":0.0151519775390625,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_waleed_W_g/std":0.4135751515933059,"train/train/tensor_act_model_layers_75_mlp_waleed_W_u/std":0.4843750321606704,"train/train/tensor_act_model_layers_8_input_layernorm/mean":-0.004032135009765625,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/max_abs":0.000598907470703125,"train/train/tensor_act_model_layers_60_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_o_proj/max_abs":1.3671875,"train/train/layer_model_layers_49/grad/mean":1.8235615653664386e-08,"train/train/tensor_act_model_layers_88_mlp_waleed_W_g/std":0.7031250443723452,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/max_abs":0.2373046875,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/max_abs":0.236328125,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_input_layernorm/max_abs":4.90625,"train/train/tensor_param_model_layers_17_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_8/grad/norm":0.08966477050597384,"train/train/tensor_act_model_layers_93/std":6.671893708294406,"train/train/tensor_act_model_layers_21_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/norm":3.078125,"train/train/tensor_act_model_layers_25_mlp_waleed_W_u/std":0.26660337563219544,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/mean":-2.9265880584716797e-05,"train/train/tensor_grad_model_layers_10_mlp_waleed_W_u_weight/std":7.89572379383265e-05,"train/train/tensor_param_model_layers_60_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_3/grad/norm":0.13975786156841066,"train/train/tensor_param_model_layers_84_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/norm":0.007156860074645754,"train/train/tensor_grad_model_layers_3_mlp_waleed_W_u_weight/mean":2.9080547392368317e-07,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/mean":-1.3540193322114646e-07,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/std":5.403060020026558e-05,"train/train/tensor_param_model_layers_25_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_o_proj/max_abs":3.78125,"train/train/tensor_act_model_layers_25_self_attn_o_proj/mean":-0.0014219284057617188,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/std":2.6218523155294976e-05,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/norm":4.625,"train/train/tensor_act_model_layers_50_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/norm":0.021068536736634818,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/max_abs":0.0003185272216796875,"train/train/tensor_act_model_layers_71_input_layernorm/std":1.0000009980480649,"train/train/tensor_param_model_layers_41_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp_down_proj/std":0.542002016914864,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_mlp_waleed_W_g_weight/max_abs":0.0004787445068359375,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_mlp_waleed_W_g_weight/norm":4.9375,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/max_abs":0.00018787384033203125,"train/train/tensor_act_model_layers_40_mlp_waleed_W_u/max_abs":2.5,"train/train/layer_model_layers_29/act/max_abs":26.375,"train/train/layer_model_layers_57/act/std":0.8567819956980792,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/norm":0.02112045716641632,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/norm":3.34375,"train/train/tensor_param_model_layers_41_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84/std":3.4492258064266963,"train/train/tensor_act_model_layers_67_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/std":5.4072798547278236e-05,"train/train/tensor_act_model_layers_53_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/max_abs":0.00079345703125,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/norm":0.017398124551549352,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_mlp_waleed_W_u_weight/mean":8.664210326969624e-08,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/mean":7.890776032581925e-09,"train/train/tensor_param_model_layers_49_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_82/std":3.261726481605549,"train/train/tensor_act_model_layers_69_self_attn_o_proj/std":0.2209499399932727,"train/train/layer_model_layers_13/act/max_abs":26.125,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/max_abs":0.00148773193359375,"train/train/layer_model_layers_17/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/max_abs":0.000835418701171875,"train/train/layer_model_layers_12/act/mean":-0.0010477732867002487,"train/train/tensor_act_model_layers_62_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_q_proj/max_abs":5.625,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/max_abs":0.1416015625,"train/train/tensor_act_model_layers_21_mlp/norm":422.4777264688982,"train/train/tensor_act_model_layers_26_post_attention_layernorm/std":1.0000001897769655,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/std":0.0299072265625,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_29_mlp_waleed_W_u/norm":2148.934352456305,"train/train/tensor_act_model_layers_77_self_attn_q_proj/norm":6657.503252844695,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/std":0.0244140625,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_mlp_waleed/max_abs":3.125,"train/train/tensor_act_model_layers_10_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_11/grad/std":5.1803740577312574e-05,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/norm":2678.327376731261,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/mean":-3.442983143031597e-08,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_3_mlp_waleed_W_u/std":0.45507820244765956,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/norm":4.53125,"train/train/tensor_act_model_layers_29_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_mlp_waleed_W_g_weight/mean":0.00013446807861328125,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp/norm":637.9534793566776,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_18_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_56_self_attn_k_proj/norm":4824.380659042438,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_31/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_o_proj/max_abs":10.8125,"train/train/tensor_act_model_layers_52_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/mean":1.2139207683503628e-07,"train/train/tensor_act_model_layers_42_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_44/grad/max_abs":0.00104522705078125,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_mlp_waleed/mean":0.00267791748046875,"train/train/tensor_grad_model_layers_6_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_mlp_waleed_W_g_weight/mean":-0.00014495849609375,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/mean":0.0002574920654296875,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/max_abs":0.000751495361328125,"train/train/tensor_param_model_layers_81_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer_model_layers_72/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_mlp_waleed_W_u_weight/max_abs":0.000823974609375,"train/train/tensor_act_model_layers_26_mlp/std":0.07959060596245711,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_8/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_k_proj/mean":0.013275146484375,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/std":5.120548117943134e-05,"train/train/tensor_act_model_layers_49_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/max_abs":0.1572265625,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/mean":-0.0001697540283203125,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/std":0.05908203125,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_59_post_attention_layernorm/std":1.000001059298832,"train/train/tensor_param_model_layers_12_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/norm":3.65625,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn/max_abs":1.1171875,"train/train/tensor_grad_model_layers_54_mlp_waleed_W_u_weight/norm":0.014146041412349394,"train/train/tensor_act_model_layers_93_post_attention_layernorm/max_abs":6.125,"train/train/tensor_act_model_layers_25_input_layernorm/norm":5792.612548831392,"train/train/tensor_act_model_layers_83_self_attn_v_proj/std":0.5537136757327129,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/std":5.035020297203376e-05,"train/train/tensor_grad_model_layers_70_mlp_waleed_W_u_weight/mean":3.1990930438041687e-07,"train/train/tensor_act_model_layers_6/std":3.234394124523741,"train/train/tensor_act_model_layers_78_mlp/mean":0.002819061279296875,"train/train/tensor_act_model_layers_52_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_k_proj/mean":0.03472900390625,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/norm":6.15625,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/max_abs":0.1875,"train/train/tensor_act_model_layers_74_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_70_mlp_waleed_W_u_weight/max_abs":0.162109375,"train/train/tensor_act_model_layers_64_post_attention_layernorm/max_abs":5.125,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/max_abs":0.11083984375,"train/train/tensor_act_model_layers_83_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_7/act/mean":0.0008617937564849854,"train/train/tensor_act_model_layers_55_self_attn_k_proj/norm":4970.146117858078,"train/train/tensor_act_model_layers_36_self_attn_q_proj/norm":6249.982821306796,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/mean":-0.000202178955078125,"train/train/layer_model_layers_60/act/std":0.8295679855821363,"train/train/tensor_act_model_layers_84_self_attn_o_proj/mean":0.020172119140625,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/max_abs":0.000316619873046875,"train/train/tensor_grad_model_layers_43_mlp_waleed_W_g_weight/norm":0.01335195485895116,"train/train/layer__model_layers_16/param/std":0.04755327587967154,"train/train/tensor_act_model_layers_72_self_attn_q_proj/mean":0.053466796875,"train/train/tensor_act_model_layers_64_self_attn/norm":1628.3842667660665,"train/train/tensor_param_model_layers_72_mlp_waleed_W_g_weight/max_abs":0.169921875,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/max_abs":0.001495361328125,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/std":5.4682816714552056e-05,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/max_abs":0.251953125,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_post_attention_layernorm/norm":5792.604858399233,"train/train/tensor_act_model_layers_37_mlp_waleed_W_u/norm":2427.325943712923,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/norm":0.01587078379413581,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/mean":0.000194549560546875,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/norm":0.005395206598357761,"train/train/tensor_grad_model_layers_24_mlp_waleed_W_g_weight/norm":0.015721144514680083,"train/train/layer_model_layers_92/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/max_abs":0.0001392364501953125,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/max_abs":0.12158203125,"train/train/tensor_act_model_layers_34_self_attn_q_proj/norm":5920.086686083966,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_waleed/mean":0.0014896392822265625,"train/train/tensor_act_model_layers_81_self_attn_q_proj/max_abs":7.78125,"train/train/tensor_act_model_layers_66_self_attn_q_proj/mean":0.0411376953125,"train/train/tensor_act_model_layers_39_mlp_waleed/max_abs":3.59375,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/norm":0.015520251351737896,"train/train/tensor_act_model_layers_10_input_layernorm/mean":-0.006317138671875,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/max_abs":0.0006103515625,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/max_abs":0.00013637542724609375,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/max_abs":0.1318359375,"train/train/tensor_param_model_layers_89_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_92_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/mean":-2.3096799850463867e-05,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/norm":4.25,"train/train/tensor_grad_model_layers_83_mlp_waleed_W_u_weight/max_abs":0.000640869140625,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/norm":3.078125,"train/train/layer_model_layers_65/grad/mean":3.3797129593475745e-08,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_waleed_W_g_weight/norm":4.84375,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_mlp/mean":-0.02886962890625,"train/train/tensor_act_model_layers_29_self_attn_v_proj/max_abs":2.625,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/max_abs":0.25,"train/train/layer_model_layers_25/grad/norm":0.03576766817476572,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/max_abs":0.0011444091796875,"train/train/tensor_act_model_layers_23_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_post_attention_layernorm/norm":5792.604492193705,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_23_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/max_abs":0.0006103515625,"train/train/tensor_act_model_layers_28_mlp_waleed_W_u/norm":2128.4585867613005,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_o_proj/std":0.07605020564248745,"train/train/tensor_act_model_layers_9_input_layernorm/mean":-0.00469207763671875,"train/train/tensor_act_model_layers_51_mlp_waleed/std":0.12817453157141068,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/max_abs":0.000789642333984375,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_mlp_waleed_W_u_weight/std":5.465941853337633e-05,"train/train/tensor_act_model_layers_74_mlp_waleed/norm":1791.770848743227,"train/train/tensor_act_model_layers_79_mlp/mean":-0.0009250640869140625,"train/train/tensor_act_model_layers_68_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/norm":0.013937648890003663,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_92/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_68_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_37_mlp/mean":0.00150299072265625,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/norm":0.014267906117308327,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_mlp_waleed_W_u/std":0.26220841926729227,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_input_layernorm/max_abs":5.03125,"train/train/tensor_act_model_layers_91_self_attn_o_proj/max_abs":8.1875,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/max_abs":0.19140625,"train/train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/norm":0.022536771714074255,"train/train/layer_model_layers_59/act/mean":-0.0030201077461242676,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp_waleed_W_g/mean":0.00432586669921875,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/max_abs":0.23046875,"train/train/tensor_grad_model_layers_60_mlp_waleed_W_g_weight/std":4.6077001592417834e-05,"train/train/layer__model_layers_58/param/std":0.05254118901162989,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/max_abs":0.0008087158203125,"train/train/tensor_act_model_layers_88_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_81/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn/mean":0.01806640625,"train/train/tensor_grad_model_layers_73_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/norm":4.75,"train/train/tensor_grad_model_layers_36_mlp_waleed_W_g_weight/std":3.9283230015840064e-05,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_waleed_W_g_weight/std":0.023193359375,"train/train/tensor_grad_model_layers_49_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer__model_layers_24/param/norm":19.48076600497591,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/max_abs":0.002227783203125,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/max_abs":0.000339508056640625,"train/train/layer_model_layers_8/act/max_abs":26.625,"train/train/tensor_act_model_layers_29_self_attn_q_proj/std":1.0390625448155215,"train/train/tensor_grad_model_layers_2_mlp_waleed_W_u_weight/std":0.00021562638160823748,"train/train/layer__model_layers_81/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_mlp/norm":817.3820887647857,"train/train/tensor_grad_model_layers_30_mlp_waleed_W_g_weight/norm":0.013246597635938504,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/max_abs":0.1650390625,"train/train/tensor_act_model_layers_73_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn_o_proj/mean":-0.00327301025390625,"train/train/tensor_act_model_layers_15_mlp/max_abs":0.49609375,"train/train/tensor_act_model_layers_45_mlp_waleed_W_u/norm":2540.840910790716,"train/train/tensor_act_model_layers_93_mlp_waleed_W_g/norm":9411.055979021758,"train/train/tensor_act_model_layers_43_self_attn_q_proj/mean":-0.03955078125,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/std":8.479226293007571e-05,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/mean":-1.1965632438659668e-05,"train/train/tensor_grad_model_layers_39_mlp_waleed_W_g_weight/max_abs":0.0005035400390625,"train/train/tensor_act_model_layers_6_mlp/mean":-0.00334930419921875,"train/train/tensor_act_model_layers_78_mlp_down_proj/max_abs":1.671875,"train/train/tensor_grad_model_layers_32_mlp_waleed_W_u_weight/mean":1.7233105609193444e-08,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_v_proj/norm":2491.2929846192906,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/std":4.519948864925305e-05,"train/train/tensor_act_model_layers_23_mlp_waleed/mean":-0.00165557861328125,"train/train/layer_model_layers_35/act/max_abs":25.875,"train/train/layer__model_layers_73/param/norm":22.521539169481734,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_input_layernorm/std":1.0000003551997958,"train/train/tensor_act_model_layers_15_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/std":3.1839071578710804e-05,"train/train/tensor_grad_model_layers_59_mlp_waleed_W_g_weight/std":4.5243026515565866e-05,"train/train/layer_model_layers_55/act/max_abs":23.75,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/std":0.055419921875,"train/train/tensor_grad_model_layers_89_mlp_waleed_W_u_weight/max_abs":0.00128936767578125,"train/train/tensor_act_model_layers_12_post_attention_layernorm/norm":5792.609985356043,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/mean":3.147125244140625e-05,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/norm":0.01736651310059616,"train/train/layer__model_layers_46/param/max_abs":1,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_down_proj/max_abs":3.953125,"train/train/tensor_act_model_layers_7_self_attn_q_proj/norm":7405.927125212165,"train/train/tensor_act_model_layers_65_self_attn/max_abs":1.1484375,"train/train/tensor_act_model_layers_60_mlp_waleed/std":0.13525479279557984,"train/train/tensor_param_model_layers_35_mlp_waleed_W_g_weight/norm":4.5625,"train/train/tensor_act_model_layers_54/mean":0.005771636962890625,"train/train/tensor_grad_model_layers_31_mlp_waleed_W_u_weight/mean":5.1957613322883844e-08,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/mean":-3.981590270996094e-05,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/std":0.05859375,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/std":2.5317602333177694e-05,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/norm":0.021688274023441572,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_58/grad/mean":-3.720949695978819e-08,"train/train/tensor_act_model_layers_56_mlp/max_abs":0.76953125,"train/train/tensor_param_model_layers_29_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_87_self_attn_v_proj/max_abs":5.28125,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_76/grad/norm":0.06304749662075561,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88/std":3.6875014014271095,"train/train/tensor_act_model_layers_53_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_mlp_waleed_W_u_weight/std":4.2201881975338105e-05,"train/train/layer_model_layers_58/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_mlp_waleed_W_g_weight/max_abs":0.00101470947265625,"train/train/tensor_act_model_layers_62_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_mlp_waleed_W_g/std":0.28906250674579587,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp_waleed_W_u/norm":5322.373382624707,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/mean":0.00011968612670898438,"train/train/tensor_act_model_layers_17_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46/mean":0.0105438232421875,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/mean":-2.6971101760864258e-05,"train/train/tensor_act_model_layers_93_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/mean":-1.30385160446167e-08,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/mean":-2.6941299438476562e-05,"train/train/layer__model_layers_49/param/max_abs":1,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_25/act/mean":-0.005817115306854248,"train/train/tensor_param_model_layers_76_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_down_proj/mean":-0.002094268798828125,"train/train/tensor_act_model_layers_50_mlp_waleed/frac_near_user_limit":0,"train/train/layer_model_layers_81/act/mean":0.0075081586837768555,"train/train/layer_model_layers_71/grad/mean":-1.8776238601009672e-07,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/std":6.409264786454517e-05,"train/train/tensor_act_model_layers_38_mlp_waleed_W_u/mean":0.0126800537109375,"train/train/tensor_param_model_layers_13_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/norm":0.0021558811461394292,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_o_proj/max_abs":0.90625,"train/train/tensor_param_model_layers_58_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn/max_abs":3.203125,"train/train/tensor_act_model_layers_93_self_attn_o_proj/max_abs":6.75,"train/train/tensor_act_model_layers_3_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_k_proj/max_abs":4.8125,"train/train/tensor_act_model_layers_58_input_layernorm/mean":0.0018011033535003662,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70/frac_near_dtype_limit":0,"train/train/layer_model_layers_72/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_o_proj/max_abs":2.09375,"train/train/tensor_act_model_layers_64_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_k_proj/std":0.9091814432424276,"train/train/tensor_act_model_layers_10_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_mlp_waleed_W_g_weight/norm":0.016663827256886428,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_v_proj/norm":2512.36952942661,"train/train/tensor_param_model_layers_73_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/std":0.0001503077525020202,"train/train/tensor_param_model_layers_91_mlp_waleed_W_g_weight/mean":6.67572021484375e-05,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/std":7.089763131313521e-05,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/max_abs":0.279296875,"train/train/tensor_act_model_layers_82_self_attn_o_proj/std":0.3994216229827387,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/max_abs":0.000858306884765625,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/std":0.00010325985923435563,"train/train/tensor_param_model_layers_3_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14/max_abs":26.125,"train/train/tensor_grad_model_layers_79_mlp_waleed_W_g_weight/mean":-3.475579433143139e-07,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/mean":2.9802322387695312e-05,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/norm":0.0007795831101148359,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/std":3.325894118077813e-05,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/std":0.000113254137623118,"train/train/tensor_act_model_layers_65_mlp/norm":619.7475039864526,"train/train/tensor_act_model_layers_90_input_layernorm/std":1.0000001438893276,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/std":0.037841796875,"train/train/tensor_grad_model_layers_40_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_mlp_waleed_W_g_weight/std":3.570493072506369e-05,"train/train/tensor_act_model_layers_5_self_attn_o_proj/max_abs":1.109375,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/max_abs":0.0008697509765625,"train/train/tensor_act_model_layers_62_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_50/grad/norm":0.03888488890527489,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/mean":2.187490463256836e-05,"train/train/tensor_act_model_layers_63_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_mlp_waleed_W_u_weight/max_abs":0.00152587890625,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/norm":0.019230433378588072,"train/train/tensor_grad_model_layers_68_mlp_waleed_W_g_weight/mean":-2.444721758365631e-07,"train/train/tensor_grad_model_layers_54_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn/norm":624.365495066896,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_mlp/std":0.11413597347699378,"train/train/tensor_act_model_layers_27_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/layer__model_layers_23/param/max_abs":1,"train/train/tensor_act_model_layers_8_mlp_down_proj/max_abs":0.74609375,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/norm":5.75,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/norm":0.029969691224975295,"train/train/tensor_act_model_layers_33_self_attn/std":0.0572521984113172,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/max_abs":0.00086212158203125,"train/train/tensor_grad_model_layers_5_mlp_waleed_W_u_weight/norm":0.03498267664669367,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/mean":-4.4405460357666016e-06,"train/train/layer_model_layers_85/grad/mean":1.9589538559573303e-07,"train/train/tensor_act_model_layers_84_mlp_waleed/max_abs":8.125,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/norm":4.3125,"train/train/tensor_act_model_layers_59_self_attn_q_proj/max_abs":5.21875,"train/train/tensor_act_model_layers_61_self_attn_k_proj/std":1.1289130832702903,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/max_abs":3.359375,"train/train/tensor_grad_model_layers_91_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/mean":-4.029273986816406e-05,"train/train/tensor_act_model_layers_9_self_attn/mean":-0.001117706298828125,"train/train/tensor_act_model_layers_19_mlp_waleed_W_u/max_abs":2.015625,"train/train/tensor_act_model_layers_35_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn/max_abs":2.59375,"train/train/tensor_act_model_layers_47_self_attn/norm":160.16051710661873,"train/train/tensor_act_model_layers_82_self_attn_o_proj/norm":2313.196253718315,"train/train/tensor_act_model_layers_44_self_attn_q_proj/mean":-0.0002808570861816406,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/std":6.80157079607826e-05,"train/train/layer_model_layers_88/act/norm":27018.48793607294,"train/train/tensor_act_model_layers_1_input_layernorm/std":1.0000000339932733,"train/train/tensor_act_model_layers_69_mlp_waleed/std":0.17675784161025984,"train/train/tensor_grad_model_layers_12_mlp_waleed_W_u_weight/std":6.678956306323898e-05,"train/train/tensor_act_model_layers_16_mlp_waleed/std":0.08203125283831637,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/mean":4.220008850097656e-05,"train/train/tensor_act_model_layers_80/mean":0.03509521484375,"train/train/tensor_grad_model_layers_69_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_mlp_waleed_W_g_weight/std":3.964155899432591e-05,"train/train/tensor_act_model_layers_93_self_attn_k_proj/mean":0.01385498046875,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/norm":0.010092555737189144,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/max_abs":0.00018024444580078125,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_61/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/mean":-4.307366907596588e-08,"train/train/tensor_act_model_layers_50_post_attention_layernorm/max_abs":5.15625,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/mean":0.00014400482177734375,"train/train/tensor_param_model_layers_21_mlp_waleed_W_u_weight/max_abs":0.1142578125,"train/train/tensor_act_model_layers_73_mlp_waleed_W_u/max_abs":2.875,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/max_abs":0.002960205078125,"train/train/tensor_act_model_layers_14_self_attn_k_proj/norm":5314.3260636629475,"train/train/tensor_grad_model_layers_33_mlp_waleed_W_u_weight/norm":0.012512728052223726,"train/train/tensor_act_model_layers_2/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/norm":4.84375,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/norm":6.53125,"train/train/tensor_act_model_layers_4_input_layernorm/max_abs":4.53125,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_mlp_waleed_W_g/mean":0.0113525390625,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/mean":3.833883965853602e-08,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/norm":0.015488054242581092,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/max_abs":0.11474609375,"train/train/layer__model_layers_48/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_post_attention_layernorm/norm":5792.614013673547,"train/train/layer__model_layers_78/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_post_attention_layernorm/mean":0.014862060546875,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/max_abs":0.130859375,"train/train/layer__model_layers_38/param/mean":0.001597292895622075,"train/train/tensor_act_model_layers_45_self_attn_v_proj/norm":2364.3232490895225,"train/train/tensor_act_model_layers_83_self_attn_v_proj/max_abs":4.25,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/max_abs":0.0001964569091796875,"train/train/tensor_act_model_layers_86_self_attn_o_proj/std":0.3530609017134962,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/std":1.8726969018579336e-05,"train/train/tensor_act_model_layers_79_post_attention_layernorm/mean":0.00699615478515625,"train/train/tensor_act_model_layers_30/norm":15585.40138721322,"train/train/layer_model_layers_88/act/max_abs":33,"train/train/tensor_grad_model_layers_88_mlp_waleed_W_g_weight/norm":0.03400069097209489,"train/train/tensor_act_model_layers_10_self_attn_o_proj/norm":976.183720747352,"train/train/tensor_param_model_layers_40_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_mlp_waleed_W_u_weight/std":3.600319682469163e-05,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/std":0.028076171875,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/norm":3.328125,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_input_layernorm/norm":5792.6131591820895,"train/train/tensor_act_model_layers_66_mlp_waleed/norm":1327.5158484257918,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/mean":8.995993994176388e-08,"train/train/tensor_act_model_layers_23_self_attn_k_proj/max_abs":4.9375,"train/train/tensor_act_model_layers_19_self_attn_k_proj/std":1.0371149943221063,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/norm":0.01773646354980655,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/std":1.5110486579050754e-05,"train/train/tensor_grad_model_layers_77_mlp_waleed_W_g_weight/max_abs":0.000698089599609375,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/norm":0.007379845143972033,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/norm":5.625,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/std":0.037109375,"train/train/tensor_act_model_layers_53_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_input_layernorm/norm":5792.610839851786,"train/train/tensor_act_model_layers_53/max_abs":23.625,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/max_abs":0.00075531005859375,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/std":0.0302734375,"train/train/tensor_act_model_layers_55_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/std":0.0242919921875,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/std":3.2850830751009114e-05,"train/train/tensor_grad_model_layers_22_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn/std":0.0977816773878899,"train/train/tensor_act_model_layers_81_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/norm":3.3125,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/max_abs":0.0003376007080078125,"train/train/tensor_param_model_layers_51_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp_waleed_W_g/norm":2123.9916451794597,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/max_abs":0.0004444122314453125,"train/train/tensor_act_model_layers_72/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_o_proj/max_abs":2.984375,"train/train/layer_model_layers_4/grad/std":0.00012308027494107245,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/max_abs":0.000194549560546875,"train/train/tensor_param_model_layers_84_mlp_waleed_W_u_weight/mean":0.00013065338134765625,"train/train/tensor_act_model_layers_10_mlp/std":0.11169509005992263,"train/train/tensor_act_model_layers_8_mlp_waleed/norm":1062.0288491334597,"train/train/tensor_act_model_layers_85_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/norm":3.125,"train/train/tensor_act_model_layers_92_mlp/max_abs":12.1875,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/mean":4.190951585769653e-08,"train/train/layer__model_layers_82/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_80/act/max_abs":26.875,"train/train/tensor_param_model_layers_74_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_v_proj/std":0.5732447122131586,"train/train/tensor_act_model_layers_86/norm":20215.525526627593,"train/train/tensor_act_model_layers_74_input_layernorm/mean":-0.000270843505859375,"train/train/tensor_act_model_layers_37_self_attn/norm":629.7152131728639,"train/train/layer__model_layers_43/param/norm":19.882503988274465,"train/train/tensor_act_model_layers_2_mlp/norm":3445.2096531453913,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_79_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_63/act/norm":19796.865143277486,"train/train/layer__model_layers_41/param/norm":19.508668195590005,"train/train/tensor_act_model_layers_80_post_attention_layernorm/norm":5792.60583496342,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/norm":3.953125,"train/train/tensor_param_model_layers_61_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/std":0.0001302743134021494,"train/train/tensor_act_model_layers_92_self_attn_q_proj/max_abs":6.1875,"train/train/tensor_act_model_layers_35_post_attention_layernorm/std":1.00000054463984,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_78_mlp_waleed_W_g/mean":-0.002964019775390625,"train/train/layer__model_layers_60/param/std":0.05045885787599827,"train/train/tensor_act_model_layers_81_self_attn/mean":0.0156402587890625,"train/train/tensor_act_model_layers_50_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_88/act/mean":-0.001554332673549652,"train/train/tensor_act_model_layers_13_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn_v_proj/std":0.38085941005593527,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/max_abs":0.1474609375,"train/train/tensor_act_model_layers_8_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89/std":3.8281255206282903,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/mean":0.00014495849609375,"train/train/tensor_act_model_layers_50_self_attn_k_proj/norm":5466.121633339059,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/std":0.025390625,"train/train/tensor_act_model_layers_11_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_0/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_post_attention_layernorm/max_abs":5.125,"train/train/tensor_act_model_layers_7_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/std":5.401317621747028e-05,"train/train/layer_model_layers_22/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_input_layernorm/max_abs":5.0625,"train/train/tensor_param_model_layers_52_mlp_waleed_W_g_weight/norm":5.09375,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/max_abs":0.1455078125,"train/train/tensor_act_model_layers_18_self_attn_o_proj/mean":0.00010704994201660156,"train/train/tensor_act_model_layers_37_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/max_abs":0.0001964569091796875,"train/train/tensor_param_model_layers_80_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/std":0.0001675409832794089,"train/train/tensor_act_model_layers_83_self_attn_o_proj/norm":1935.1022345329607,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/max_abs":0.000919342041015625,"train/train/tensor_act_model_layers_6_input_layernorm/std":1.0000001071020903,"train/train/tensor_param_model_layers_13_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/std":5.6917327990176615e-05,"train/train/tensor_act_model_layers_65_mlp_down_proj/mean":0.0016021728515625,"train/train/tensor_act_model_layers_42_self_attn_v_proj/norm":2203.6200992830013,"train/train/tensor_act_model_layers_28_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_32/grad/mean":1.3112615032612627e-09,"train/train/tensor_param_model_layers_1_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_v_proj/std":0.37304708096363715,"train/train/tensor_grad_model_layers_84_mlp_waleed_W_g_weight/norm":0.03137278277705096,"train/train/tensor_act_model_layers_20_self_attn_k_proj/mean":0.005615234375,"train/train/tensor_act_model_layers_51_self_attn/max_abs":3.71875,"train/train/tensor_act_model_layers_49_mlp_waleed_W_g/std":0.35986433197240764,"train/train/tensor_act_model_layers_67_self_attn_q_proj/mean":0.0628662109375,"train/train/layer__model_layers_14/param/norm":19.125012765518616,"train/train/tensor_act_model_layers_50_self_attn/mean":0.0011072158813476562,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/mean":-4.547834396362305e-05,"train/train/tensor_act_model_layers_85/norm":19867.99815502181,"train/train/tensor_grad_model_layers_28_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_mlp_waleed_W_g_weight/mean":4.647299647331238e-07,"train/train/tensor_act_model_layers_15_mlp_down_proj/norm":383.73028644989967,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_waleed_W_g_weight/norm":0.01596171446938294,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/norm":0.019536399162553394,"train/train/tensor_act_model_layers_58_mlp_waleed_W_u/norm":2945.3774321446576,"train/train/tensor_grad_model_layers_48_mlp_waleed_W_u_weight/mean":-6.647314876317978e-08,"train/train/tensor_act_model_layers_78_post_attention_layernorm/norm":5792.609008790065,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/max_abs":0.000942230224609375,"train/train/layer__model_layers_83/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/norm":0.016053290387549785,"train/train/tensor_act_model_layers_71_mlp_down_proj/max_abs":1.0625,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/max_abs":0.1826171875,"train/train/tensor_act_model_layers_55_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/max_abs":0.000919342041015625,"train/train/tensor_act_model_layers_56_mlp_waleed/std":0.12792968977498642,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_60_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/norm":0.00346913527484108,"train/train/tensor_act_model_layers_79_post_attention_layernorm/max_abs":5.46875,"train/train/tensor_param_model_layers_19_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_70_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/std":0.049560546875,"train/train/tensor_act_model_layers_63_mlp_waleed/std":0.15722659396828756,"train/train/tensor_param_model_layers_54_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_mlp_waleed_W_g_weight/norm":12.1875,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/mean":-0.0003814697265625,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/std":7.055507903159906e-05,"train/train/layer_model_layers_28/act/norm":20570.237618139574,"train/train/tensor_param_model_layers_16_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_31_mlp_waleed_W_g/std":0.2832031295988066,"train/train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_37/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp_down_proj/norm":429.6958412000396,"train/train/tensor_act_model_layers_41_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_q_proj/norm":7275.766697128534,"train/train/tensor_grad_model_layers_0_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_down_proj/norm":2591.82479131875,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/std":3.7515576365014325e-05,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_mlp/norm":587.8200903843738,"train/train/tensor_act_model_layers_38_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/std":0.045166015625,"train/train/layer_model_layers_49/act/max_abs":23.875,"train/train/tensor_act_model_layers_38_self_attn_v_proj/std":0.5048856649258276,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/mean":-3.2712705433368683e-07,"train/train/tensor_act_model_layers_91_input_layernorm/max_abs":5.65625,"train/train/tensor_act_model_layers_61_mlp_waleed/mean":-0.002391815185546875,"train/train/tensor_act_model_layers_49_input_layernorm/std":1.0000011479471829,"train/train/tensor_act_model_layers_74_self_attn_q_proj/max_abs":6.71875,"train/train/tensor_act_model_layers_82/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/max_abs":0.02587890625,"train/train/tensor_act_model_layers_7_self_attn/std":0.09131191012946183,"train/train/tensor_param_model_layers_77_mlp_waleed_W_g_weight/std":0.037109375,"train/train/layer_model_layers_73/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/norm":0.01865170862205031,"train/train/tensor_act_model_layers_41_mlp_waleed/norm":845.1655378253636,"train/train/tensor_act_model_layers_39_input_layernorm/max_abs":5.28125,"train/train/tensor_param_model_layers_88_mlp_waleed_W_g_weight/std":0.048583984375,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/max_abs":0.1279296875,"train/train/tensor_act_model_layers_8_mlp_down_proj/mean":-0.00044536590576171875,"train/train/tensor_act_model_layers_31/frac_near_user_limit":0,"train/train/global/param/norm":213.04340103053238,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/std":0.04443359375,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/mean":-1.276843249797821e-06,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/mean":-1.8122955225408077e-07,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/norm":0.013387341868021405,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/mean":6.77257776260376e-06,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_mlp_waleed_W_u/max_abs":2.296875,"train/train/tensor_grad_model_layers_61_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_k_proj/std":0.8017607471340338,"train/train/tensor_act_model_layers_35_self_attn_q_proj/mean":-0.03350830078125,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/std":4.7481221521818374e-05,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/std":0.00102996826171875,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/norm":5476.762485378507,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn/mean":0.00024271011352539062,"train/train/tensor_act_model_layers_91_self_attn_q_proj/std":1.2226624900000511,"train/train/tensor_act_model_layers_64_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp/max_abs":3.359375,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/norm":0.013428636000446829,"train/train/tensor_act_model_layers_57_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/max_abs":0.00092315673828125,"train/train/tensor_act_model_layers_20_input_layernorm/max_abs":5.21875,"train/train/tensor_act_model_layers_72_mlp/mean":-0.000789642333984375,"train/train/tensor_act_model_layers_54_self_attn_q_proj/max_abs":6.1875,"train/train/tensor_param_model_layers_50_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp_waleed_W_u/mean":-0.002307891845703125,"train/train/tensor_act_model_layers_2_mlp_down_proj/mean":-0.00658416748046875,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_1_mlp_waleed_W_u_weight/mean":-3.7834979593753815e-07,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/norm":5.75,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/mean":7.200241088867188e-05,"train/train/tensor_act_model_layers_15_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_62_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp/mean":-0.0003304481506347656,"train/train/tensor_act_model_layers_87/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_waleed_W_u_weight/max_abs":0.0010223388671875,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_user_limit":0,"_wandb":{"runtime":3875},"train/train/tensor_act_model_layers_76_mlp_waleed/max_abs":4.75,"train/train/tensor_act_model_layers_33_self_attn/norm":331.91591208265316,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/std":2.905605764321166e-05,"train/train/tensor_act_model_layers_62_self_attn/mean":0.0013713836669921875,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/std":0.0311279296875,"train/train/tensor_act_model_layers_2_mlp_waleed_W_g/norm":4059.652438444562,"train/train/tensor_param_model_layers_60_mlp_waleed_W_u_weight/norm":5.25,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/mean":1.2223608791828156e-09,"train/train/tensor_act_model_layers_50_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_waleed_W_u_weight/norm":4.78125,"train/train/tensor_act_model_layers_83_mlp_waleed_W_g/std":0.5664064769086712,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/std":5.0865403485237695e-05,"train/train/tensor_act_model_layers_1/mean":-0.00400543212890625,"train/train/layer_model_layers_34/act/std":0.8558940953634442,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/std":7.285906670810677e-05,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/std":0.0233154296875,"train/train/layer_model_layers_25/act/norm":20270.057365339184,"train/train/tensor_grad_model_layers_62_mlp_waleed_W_u_weight/max_abs":0.0007781982421875,"train/train/tensor_act_model_layers_3_self_attn_v_proj/max_abs":2.125,"train/train/tensor_grad_model_layers_16_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_mlp_waleed_W_u_weight/mean":-5.279434844851494e-08,"train/train/layer_model_layers_30/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_70/act/max_abs":25.25,"train/train/tensor_act_model_layers_81_input_layernorm/norm":5792.609497076923,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/norm":5.1875,"train/train/tensor_act_model_layers_49_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn/std":0.2470715855833901,"train/train/tensor_act_model_layers_82_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_77_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/mean":-0.00010061264038085938,"train/train/tensor_act_model_layers_86_self_attn_q_proj/max_abs":7.09375,"train/train/tensor_act_model_layers_64_input_layernorm/max_abs":5.125,"train/train/tensor_param_model_layers_70_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_30_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/max_abs":0.000537872314453125,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/max_abs":0.09521484375,"train/train/layer_model_layers_13/grad/max_abs":0.00372314453125,"train/train/tensor_act_model_layers_49_post_attention_layernorm/norm":5792.610717774687,"train/train/tensor_act_model_layers_38_input_layernorm/norm":5792.611450197138,"train/train/tensor_act_model_layers_34_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_mlp_waleed_W_u_weight/std":4.7089529694990475e-05,"train/train/tensor_act_model_layers_56_mlp/norm":480.87008708016845,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/norm":0.004662228051006761,"train/train/tensor_act_model_layers_0_mlp/norm":13490.920508464298,"train/train/tensor_act_model_layers_26_self_attn/norm":440.58534308160813,"train/train/tensor_act_model_layers_29_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/mean":0.0002460479736328125,"train/train/tensor_grad_model_layers_85_mlp_waleed_W_g_weight/max_abs":0.000762939453125,"train/train/layer_model_layers_42/grad/mean":-1.1690205047357473e-08,"train/train/tensor_act_model_layers_43_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_mlp/std":0.06384312567946486,"train/train/layer_model_layers_81/grad/frac_near_user_limit":0,"train/train/layer_model_layers_80/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_waleed_W_u/std":0.431640703652502,"train/train/layer__model_layers_78/param/mean":0.001626001319349649,"train/train/tensor_act_model_layers_73/frac_near_dtype_limit":0,"train/train/layer_model_layers_52/grad/std":3.7086642103721176e-05,"train/train/tensor_act_model_layers_33_mlp/mean":0.00066375732421875,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/mean":-4.6253204345703125e-05,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_mlp_waleed_W_u_weight/max_abs":0.0004425048828125,"train/train/tensor_act_model_layers_33_mlp/std":0.04638704720180074,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/norm":0.00952871976943452,"train/train/tensor_act_model_layers_64_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/norm":5.625,"train/train/tensor_param_model_layers_38_mlp_waleed_W_u_weight/mean":4.38690185546875e-05,"train/train/tensor_act_model_layers_10_self_attn/mean":0.0006036758422851562,"train/train/layer__model_layers_0/param/mean":0.001651805574176092,"train/train/tensor_act_model_layers_27_mlp/norm":391.17678248616306,"train/train/tensor_act_model_layers_23_mlp/max_abs":0.439453125,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/norm":0.009797521001029668,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/norm":0.0006125945362690877,"train/train/tensor_act_model_layers_63_mlp_waleed_W_u/max_abs":2.59375,"train/train/tensor_act_model_layers_10/max_abs":26.25,"train/train/tensor_act_model_layers_68_mlp_waleed/frac_near_user_limit":0,"train/train/layer_model_layers_6/grad/mean":-2.224916554464379e-07,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/mean":-1.6868580132722855e-07,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/norm":0.009603974445237124,"train/train/tensor_param_model_layers_37_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp/max_abs":0.76171875,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/norm":0.015306109486484391,"train/train/tensor_act_model_layers_66_self_attn_q_proj/max_abs":5.625,"train/train/layer__model_layers_66/param/norm":21.598271957786345,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/std":1.4758651526291896e-05,"train/train/layer__model_layers_67/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_74/act/max_abs":25.875,"train/train/tensor_param_model_layers_54_mlp_waleed_W_g_weight/std":0.02783203125,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_70/act/norm":20839.207675769485,"train/train/tensor_act_model_layers_9_input_layernorm/std":1.0000000766885904,"train/train/tensor_act_model_layers_45_mlp_waleed/mean":0.0010023117065429688,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/mean":0.0001163482666015625,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/mean":-1.0174699127674103e-07,"train/train/tensor_param_model_layers_59_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_3_mlp_waleed_W_u_weight/norm":4.40625,"train/train/tensor_act_model_layers_79_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp_down_proj/std":0.07287628516813972,"train/train/tensor_act_model_layers_2_input_layernorm/max_abs":4.25,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/std":0.00010252726867990716,"train/train/tensor_act_model_layers_87_self_attn_q_proj/norm":7377.232894413768,"train/train/tensor_act_model_layers_5_self_attn_q_proj/mean":0.03790283203125,"train/train/tensor_act_model_layers_61_input_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_75/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_v_proj/mean":-0.00371551513671875,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/std":0.714843766733271,"train/train/tensor_act_model_layers_35_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_mlp_waleed_W_u_weight/max_abs":0.0006866455078125,"train/train/tensor_act_model_layers_72_mlp_waleed/max_abs":4.3125,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/max_abs":0.232421875,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/max_abs":0.2158203125,"train/train/tensor_param_model_layers_18_mlp_waleed_W_g_weight/max_abs":0.10693359375,"train/train/tensor_act_model_layers_62_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn/norm":671.6598145305732,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/max_abs":0.0009765625,"train/train/tensor_act_model_layers_33_mlp_waleed/max_abs":2.03125,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/std":0.045166015625,"train/train/tensor_act_model_layers_16_mlp/mean":-0.0002503395080566406,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/std":0.022705078125,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_78_input_layernorm/norm":5792.61059570492,"train/train/layer__model_layers_56/param/norm":20.45748735946084,"train/train/tensor_act_model_layers_11_mlp_waleed_W_g/max_abs":2.328125,"train/train/tensor_act_model_layers_2/max_abs":27.5,"train/train/tensor_param_model_layers_71_mlp_waleed_W_u_weight/max_abs":0.2060546875,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn_o_proj/norm":2277.4074541225787,"train/train/tensor_act_model_layers_0_self_attn_o_proj/std":0.6455159689969379,"train/train/tensor_act_model_layers_40_mlp_down_proj/max_abs":0.5703125,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_o_proj/max_abs":2.59375,"train/train/tensor_param_model_layers_8_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn/max_abs":0.828125,"train/train/tensor_act_model_layers_26/norm":15869.476397611805,"train/train/layer_model_layers_8/act/std":1.1256397711939896,"train/train/tensor_param_model_layers_18_mlp_waleed_W_u_weight/std":0.023193359375,"train/train/tensor_act_model_layers_92_self_attn_o_proj/norm":3668.9699777493547,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_49/param/norm":19.554024222170995,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/mean":-0.00020599365234375,"train/train/tensor_grad_model_layers_56_mlp_waleed_W_g_weight/mean":-3.7834979593753815e-09,"train/train/layer_model_layers_71/act/mean":0.004364490509033203,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/max_abs":0.1572265625,"train/train/layer__model_layers_8/param/mean":0.0015429468497099258,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/mean":5.1979441195726395e-08,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/max_abs":0.150390625,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83/norm":19298.35232351818,"train/train/tensor_act_model_layers_19_mlp_waleed_W_g/mean":0.0102996826171875,"train/train/layer__model_layers_82/param/mean":0.0015123578575955538,"train/train/tensor_act_model_layers_53_input_layernorm/max_abs":5,"train/train/tensor_act_model_layers_47_mlp/norm":370.04833947114315,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/max_abs":0.1328125,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_waleed_W_g/max_abs":2,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn/mean":-0.00016647577285766602,"train/train/tensor_param_model_layers_58_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_waleed_W_g_weight/max_abs":0.0010223388671875,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_57/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_waleed_W_g_weight/max_abs":0.1044921875,"train/train/tensor_act_model_layers_20_post_attention_layernorm/norm":5792.600463874075,"train/train/tensor_grad_model_layers_50_mlp_waleed_W_u_weight/std":4.4006490312182143e-05,"train/train/tensor_act_model_layers_36_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_mlp_waleed_W_g_weight/norm":0.014207333865442049,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/norm":3.375,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/mean":-8.96453857421875e-05,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_o_proj/std":0.17236677124180275,"train/train/tensor_act_model_layers_16_self_attn_q_proj/std":1.1738330044383158,"train/train/tensor_act_model_layers_11_self_attn_o_proj/std":0.085449952296425,"train/train/tensor_param_model_layers_25_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn_v_proj/mean":0.0023479461669921875,"train/train/tensor_act_model_layers_59_mlp_waleed/max_abs":2.734375,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_down_proj/mean":-0.00206756591796875,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/std":0.0281982421875,"train/train/layer__model_layers_18/param/mean":0.001566846731486447,"train/train/tensor_act_model_layers_67_mlp/mean":-0.00206756591796875,"train/train/tensor_act_model_layers_45_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_o_proj/norm":3511.495258268671,"train/train/tensor_act_model_layers_22_self_attn_v_proj/norm":2281.7688568847675,"train/train/tensor_act_model_layers_70_mlp_waleed/mean":0.0004553794860839844,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/std":5.466560707000503e-05,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/mean":-3.9068982005119324e-07,"train/train/tensor_grad_model_layers_50_mlp_waleed_W_g_weight/max_abs":0.000850677490234375,"train/train/tensor_param_model_layers_56_mlp_waleed_W_g_weight/max_abs":0.1494140625,"train/train/tensor_act_model_layers_82_mlp_waleed_W_g/mean":-0.0352783203125,"train/train/tensor_grad_model_layers_71_mlp_waleed_W_u_weight/mean":2.2515905584441498e-08,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/norm":0.0009417918957913279,"train/train/tensor_act_model_layers_23_input_layernorm/norm":5792.609008789848,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/norm":0.0031988732974698674,"train/train/tensor_act_model_layers_6_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_mlp_waleed_W_u_weight/norm":0.031776581176536235,"train/train/tensor_act_model_layers_47_mlp_down_proj/mean":0.0016117095947265625,"train/train/tensor_act_model_layers_7_mlp_waleed/std":0.179932140551187,"train/train/tensor_act_model_layers_62_self_attn/std":0.08752472353641778,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/norm":0.02762183068076932,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/mean":7.018446922302246e-06,"train/train/tensor_act_model_layers_19_self_attn_v_proj/norm":1904.2233784475322,"train/train/tensor_act_model_layers_9_mlp/std":0.09399446039950485,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/max_abs":0.2470703125,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/std":3.240093176797252e-05,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/norm":0.0027377835384646667,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/mean":2.2411346435546875e-05,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/mean":-3.632158041000366e-08,"train/train/tensor_param_model_layers_75_mlp_waleed_W_g_weight/std":0.036376953125,"train/train/tensor_act_model_layers_33_self_attn/max_abs":0.8984375,"train/train/layer_model_layers_23/grad/mean":1.411120786495774e-07,"train/train/tensor_act_model_layers_87/std":3.582037967411168,"train/train/tensor_param_model_layers_20_mlp_waleed_W_u_weight/max_abs":0.11376953125,"train/train/tensor_param_model_layers_14_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_63_mlp_down_proj/norm":587.1392161318835,"train/train/tensor_act_model_layers_82_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_mlp_waleed_W_g_weight/max_abs":0.12255859375,"train/train/tensor_act_model_layers_89_input_layernorm/max_abs":5.96875,"train/train/tensor_act_model_layers_60_mlp_waleed/max_abs":2.671875,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_q_proj/norm":6262.577743726729,"train/train/tensor_param_model_layers_38_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/mean":7.104873657226562e-05,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/mean":-0.000423431396484375,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/std":0.00010504417436542145,"train/train/tensor_param_model_layers_15_mlp_waleed_W_u_weight/max_abs":0.11572265625,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/max_abs":0.173828125,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_waleed/std":0.1547859293574474,"train/train/tensor_act_model_layers_16_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_mlp_waleed_W_u_weight/std":6.271416386880365e-05,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/max_abs":0.1376953125,"train/train/tensor_act_model_layers_60_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_k_proj/std":1.0742259077353862,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/mean":1.8596649169921875e-05,"train/train/tensor_param_model_layers_71_mlp_waleed_W_g_weight/mean":-0.000164031982421875,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/mean":0.00035858154296875,"train/train/tensor_grad_model_layers_27_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/max_abs":0.13671875,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89/norm":22169.49610706931,"train/train/tensor_act_model_layers_46_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_26/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7/std":3.1836203488559627,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_waleed_W_g/mean":-1.5348196029663086e-05,"train/train/tensor_act_model_layers_17_input_layernorm/std":1.0000000076834112,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/mean":5.602836608886719e-06,"train/train/tensor_act_model_layers_74_mlp_waleed_W_g/mean":0.0070953369140625,"train/train/tensor_grad_model_layers_11_mlp_waleed_W_u_weight/mean":1.4854595065116882e-07,"train/train/tensor_param_model_layers_62_mlp_waleed_W_g_weight/std":0.0306396484375,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/norm":0.009674312937537627,"train/train/tensor_grad_model_layers_43_mlp_waleed_W_u_weight/max_abs":0.000843048095703125,"train/train/tensor_param_model_layers_44_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_30_self_attn_v_proj/norm":2600.3372900300624,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/norm":0.010697227370698878,"train/train/layer_model_layers_68/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/mean":0.00016498565673828125,"train/train/tensor_act_model_layers_70_mlp_waleed_W_u/max_abs":2.75,"train/train/tensor_act_model_layers_41_self_attn_o_proj/norm":267.3777887339979,"train/train/tensor_act_model_layers_60_post_attention_layernorm/mean":0.0034427642822265625,"train/train/layer_model_layers_61/grad/std":5.2064423894955746e-05,"train/train/tensor_param_model_layers_87_mlp_waleed_W_u_weight/mean":-0.00011777877807617188,"train/train/tensor_act_model_layers_67_mlp/max_abs":1.0859375,"train/train/tensor_act_model_layers_72_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/std":0.00016852669772324984,"train/train/tensor_act_model_layers_18_self_attn_v_proj/std":0.38574356113469177,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/mean":-2.205371856689453e-05,"train/train/layer_model_layers_6/act/max_abs":26.875,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/max_abs":0.00021457672119140625,"train/train/tensor_act_model_layers_42_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_k_proj/norm":7710.032623166996,"train/train/tensor_act_model_layers_41_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/norm":0.0008416271749495729,"train/train/tensor_act_model_layers_47_mlp/mean":0.0016117095947265625,"train/train/tensor_act_model_layers_41_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_mlp/mean":0.00017058849334716797,"train/train/tensor_param_model_layers_43_mlp_waleed_W_g_weight/norm":4.75,"train/train/tensor_act_model_layers_10/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_post_attention_layernorm/std":1.0000001679872952,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/mean":2.2829044610261917e-07,"train/train/layer__model_layers_24/param/std":0.04807518955597359,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/std":8.23689129652138e-05,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/std":0.03759765625,"train/train/tensor_act_model_layers_14_self_attn_k_proj/mean":0.026580810546875,"train/train/layer_model_layers_5/act/mean":0.00011409074068069458,"train/train/tensor_grad_model_layers_55_mlp_waleed_W_g_weight/mean":-6.175832822918892e-08,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/mean":0.00010442733764648438,"train/train/layer_model_layers_44/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp_down_proj/mean":0.0005474090576171875,"train/train/tensor_act_model_layers_51_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer_model_layers_88/grad/std":8.839082435448447e-05,"train/train/tensor_grad_model_layers_27_mlp_waleed_W_u_weight/norm":0.014972068142107936,"train/train/tensor_act_model_layers_30_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/std":0.0001322243766122691,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/std":7.36719045158384e-05,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/norm":0.0036489568397877555,"train/train/tensor_param_model_layers_57_input_layernorm_weight/std":0,"train/train/layer_model_layers_89/act/mean":-0.017221271991729736,"train/train/tensor_act_model_layers_51_mlp/std":0.08105542862740756,"train/train/tensor_act_model_layers_56_self_attn_v_proj/std":0.3847682637067299,"train/train/tensor_act_model_layers_27/norm":15788.51625194786,"train/train/tensor_act_model_layers_80_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_37_mlp_waleed_W_u/max_abs":2.359375,"train/train/tensor_act_model_layers_66_post_attention_layernorm/norm":5792.617919928434,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/max_abs":0.12890625,"train/train/tensor_grad_model_layers_29_mlp_waleed_W_g_weight/norm":0.012573982883598263,"train/train/tensor_grad_model_layers_64_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/mean":-1.0558869689702988e-07,"train/train/layer__model_layers_6/param/norm":19.572768371355902,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/norm":3.328125,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/max_abs":0.00019073486328125,"train/train/tensor_act_model_layers_14_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp_waleed_W_u/std":0.2807630024756233,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/std":0.037109375,"train/train/tensor_act_model_layers_80_self_attn_o_proj/norm":3130.742206041148,"train/train/tensor_act_model_layers_71_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_mlp_waleed_W_g_weight/norm":4.375,"train/train/tensor_act_model_layers_91_self_attn_v_proj/norm":3941.97358530611,"train/train/tensor_act_model_layers_11_mlp/std":0.06073007259754281,"train/train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_mlp_waleed_W_g/norm":2495.535523784211,"train/train/tensor_act_model_layers_46_mlp_down_proj/norm":455.54596648817636,"train/train/tensor_act_model_layers_48_post_attention_layernorm/max_abs":5.21875,"train/train/layer_model_layers_75/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/std":7.952165532851339e-05,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/mean":-0.00016021728515625,"train/train/tensor_grad_model_layers_40_mlp_waleed_W_u_weight/std":3.958437756611103e-05,"train/train/tensor_act_model_layers_62/norm":15020.882943840448,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/norm":3.96875,"train/train/tensor_param_model_layers_33_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_mlp_waleed_W_u_weight/max_abs":0.00069427490234375,"train/train/tensor_act_model_layers_52_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_mlp_down_proj/mean":0.00018596649169921875,"train/train/layer__model_layers_3/param/max_abs":1,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_waleed_W_u_weight/mean":-0.0002899169921875,"train/train/tensor_act_model_layers_65_self_attn/std":0.127686253762904,"train/train/tensor_grad_model_layers_15_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_waleed/max_abs":3.921875,"train/train/layer_model_layers_20/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn/std":0.056764184063560374,"train/train/tensor_act_model_layers_30_mlp/max_abs":0.474609375,"train/train/tensor_act_model_layers_68_input_layernorm/std":1.0000010890356388,"train/train/tensor_act_model_layers_37_self_attn/max_abs":1.0390625,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/mean":9.584426879882812e-05,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/norm":7.03125,"train/train/layer_model_layers_61/grad/norm":0.042168648813359536,"train/train/tensor_act_model_layers_24/norm":16016.275013828677,"train/train/tensor_param_model_layers_10_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_59/act/std":0.8518636580900103,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_mlp_waleed_W_g_weight/max_abs":0.00098419189453125,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/max_abs":0.000705718994140625,"train/train/tensor_act_model_layers_74_mlp_waleed/max_abs":5.25,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/norm":3.828125,"train/train/tensor_act_model_layers_44_self_attn_v_proj/std":0.345703280746565,"train/train/tensor_param_model_layers_38_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_74_post_attention_layernorm/std":1.0000007600826233,"train/train/tensor_act_model_layers_50_mlp_waleed_W_g/std":0.34033310293469476,"train/train/tensor_param_model_layers_27_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_mlp/mean":-0.03179931640625,"train/train/tensor_act_model_layers_35_self_attn_o_proj/std":0.08642889780200759,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/std":0.007066616000854374,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20/mean":-0.00583648681640625,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/norm":0.0005571286672045022,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/norm":0.0047948058788308,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/mean":0.0003204345703125,"train/train/tensor_act_model_layers_60_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/norm":7.40625,"train/train/tensor_param_model_layers_7_mlp_waleed_W_u_weight/std":0.0233154296875,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/max_abs":0.00159454345703125,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/max_abs":0.32421875,"train/train/tensor_act_model_layers_60/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn/mean":0.0105743408203125,"train/train/layer_model_layers_59/act/max_abs":24,"train/train/tensor_param_model_layers_69_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_35/grad/max_abs":0.00092315673828125,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/std":4.226431706534317e-05,"train/train/tensor_act_model_layers_71_mlp_waleed_W_g/norm":3593.498901169081,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/std":0.0001247004553690477,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/norm":0.0371781765445821,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_34_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/max_abs":0.0002079010009765625,"train/train/tensor_act_model_layers_29_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn/std":0.09594770856383782,"train/train/tensor_act_model_layers_83_mlp_down_proj/std":0.3051769633270258,"train/train/tensor_act_model_layers_88_post_attention_layernorm/norm":5792.610351563251,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_85/grad/std":9.339160474329644e-05,"train/train/tensor_param_model_layers_22_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/norm":0.033333466450107534,"train/train/tensor_act_model_layers_7_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/layer_model_layers_66/act/frac_near_user_limit":0,"train/train/layer_model_layers_73/grad/mean":3.61922704113441e-09,"train/train/tensor_act_model_layers_36_input_layernorm/std":1.0000005705731574,"train/train/tensor_grad_model_layers_35_mlp_waleed_W_u_weight/std":3.802873000241111e-05,"train/train/tensor_act_model_layers_50_self_attn_k_proj/mean":0.03106689453125,"train/train/tensor_grad_model_layers_83_mlp_waleed_W_g_weight/max_abs":0.00089263916015625,"train/train/tensor_act_model_layers_53_post_attention_layernorm/mean":0.0008835792541503906,"train/train/layer__model_layers_26/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_87/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_waleed_W_g_weight/max_abs":0.0025787353515625,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/mean":2.5779008865356445e-05,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40/norm":15247.891352224317,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_mlp_waleed_W_g_weight/mean":1.528533175587654e-07,"train/train/layer_model_layers_38/act/norm":20184.585072303962,"train/train/tensor_act_model_layers_69_self_attn/norm":1280.2571859913237,"train/train/tensor_act_model_layers_45_mlp_waleed_W_g/std":0.3203125890605482,"train/train/layer__model_layers_30/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_v_proj/norm":2343.358003997624,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/norm":0.018859962043380165,"train/train/tensor_act_model_layers_14_mlp_waleed_W_u/std":0.23315468308781295,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/norm":7,"train/train/tensor_grad_model_layers_1_mlp_waleed_W_u_weight/max_abs":0.0146484375,"train/train/tensor_act_model_layers_7_self_attn_q_proj/std":1.281250008609055,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/max_abs":0.000720977783203125,"train/train/tensor_act_model_layers_23_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/max_abs":0.123046875,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/std":0.0390625,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/std":0.027587890625,"train/train/tensor_act_model_layers_76_input_layernorm/max_abs":5.53125,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/max_abs":0.00167083740234375,"train/train/tensor_act_model_layers_83_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/mean":0.00010251998901367188,"train/train/tensor_act_model_layers_35_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/std":0.0277099609375,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/std":7.465234888491188e-05,"train/train/tensor_act_model_layers_55_self_attn_v_proj/std":0.3398438925030289,"train/train/layer_model_layers_39/act/mean":-0.0008349120616912842,"train/train/tensor_act_model_layers_79_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_k_proj/norm":5577.707221146515,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/norm":4.5625,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/mean":-1.4121178537607193e-07,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/mean":-2.728775143623352e-06,"train/train/tensor_act_model_layers_20_mlp/mean":0.0015163421630859375,"train/train/tensor_act_model_layers_8_mlp_down_proj/std":0.10205253385480936,"train/train/tensor_param_model_layers_71_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/norm":0.028478444041191014,"train/train/layer_model_layers_71/act/max_abs":25.25,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/mean":3.1618401408195496e-07,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/mean":2.9616057872772217e-06,"train/train/tensor_act_model_layers_64_input_layernorm/norm":5792.608764650387,"train/train/tensor_param_model_layers_66_mlp_waleed_W_u_weight/mean":-6.079673767089844e-05,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/std":0.040771484375,"train/train/tensor_act_model_layers_31_mlp_waleed_W_u/max_abs":1.9765625,"train/train/tensor_act_model_layers_25_mlp_down_proj/max_abs":0.66796875,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/global_step":1500,"train/train/tensor_grad_model_layers_23_mlp_waleed_W_g_weight/max_abs":0.00119781494140625,"train/train/tensor_act_model_layers_39_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_waleed_W_u/max_abs":4.40625,"train/train/tensor_act_model_layers_83_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/std":8.786397259150054e-05,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/norm":0.0006723126615391956,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/mean":2.0721927285194397e-08,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/std":0.0235595703125,"train/train/tensor_act_model_layers_60_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/mean":-3.0159950256347656e-05,"train/train/tensor_param_model_layers_45_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_mlp_waleed_W_u/std":0.5068389268536143,"train/train/tensor_act_model_layers_5_mlp/max_abs":0.9375,"train/train/layer_model_layers_85/grad/norm":0.07571038418029254,"train/train/tensor_param_model_layers_61_mlp_waleed_W_u_weight/max_abs":0.1689453125,"train/train/layer_model_layers_78/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/norm":0.037376247295885894,"train/train/tensor_act_model_layers_75_self_attn_q_proj/max_abs":5.375,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/mean":-1.4741090126335621e-08,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/max_abs":0.0010986328125,"train/train/tensor_act_model_layers_24_self_attn_o_proj/norm":1274.8418644276617,"train/train/tensor_act_model_layers_40_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/mean":-0.0001964569091796875,"train/train/tensor_grad_model_layers_58_mlp_waleed_W_u_weight/mean":8.055940270423889e-08,"train/train/tensor_act_model_layers_87_mlp_down_proj/mean":-0.0056915283203125,"train/train/tensor_act_model_layers_67_mlp_waleed/mean":0.00020134449005126953,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/norm":3.09375,"train/train/tensor_param_model_layers_83_mlp_waleed_W_u_weight/max_abs":0.197265625,"train/train/tensor_act_model_layers_74_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp/norm":423.02089659152307,"train/train/tensor_param_model_layers_47_mlp_waleed_W_u_weight/max_abs":0.130859375,"train/train/tensor_act_model_layers_79_input_layernorm/std":1.0000004857719162,"train/train/tensor_act_model_layers_44_input_layernorm/norm":5792.609741213957,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/grad/max_abs":0.00106048583984375,"train/train/tensor_grad_model_layers_46_mlp_waleed_W_u_weight/max_abs":0.000553131103515625,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/max_abs":0.000904083251953125,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/norm":0.0011296013627446525,"train/train/tensor_act_model_layers_7_self_attn_k_proj/std":1.4296877982471172,"train/train/tensor_act_model_layers_65_mlp_waleed_W_u/norm":3256.552193312002,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/max_abs":0.00010728836059570312,"train/train/tensor_param_model_layers_83_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90/mean":-0.020538330078125,"train/train/layer_model_layers_7/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/std":0.050048828125,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp/norm":366.7478613308848,"train/train/tensor_grad_model_layers_40_mlp_waleed_W_g_weight/max_abs":0.00115203857421875,"train/train/tensor_act_model_layers_19_input_layernorm/std":1.0000000160653142,"train/train/tensor_act_model_layers_44_post_attention_layernorm/std":1.0000013421276979,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn_q_proj/max_abs":6.78125,"train/train/tensor_param_model_layers_39_mlp_waleed_W_u_weight/norm":4.65625,"train/train/tensor_param_model_layers_86_mlp_waleed_W_u_weight/mean":0.000194549560546875,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/std":9.500143669133371e-05,"train/train/tensor_act_model_layers_85_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/std":0.03662109375,"train/train/tensor_param_model_layers_9_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_11_self_attn_o_proj/mean":-0.0002193450927734375,"train/train/layer_model_layers_68/act/mean":-0.002548038959503174,"train/train/tensor_act_model_layers_33_mlp/norm":268.4124111021993,"train/train/tensor_act_model_layers_11_post_attention_layernorm/max_abs":5.15625,"train/train/tensor_act_model_layers_80_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_58/param/max_abs":1,"train/train/tensor_param_model_layers_77_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/norm":0.006693740501049792,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_75_mlp_waleed_W_u_weight/max_abs":0.0006866455078125,"train/train/layer_model_layers_54/act/max_abs":23.75,"train/train/tensor_param_model_layers_51_mlp_waleed_W_g_weight/max_abs":0.1474609375,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/norm":0.010864686421715697,"train/train/tensor_act_model_layers_92_mlp_waleed_W_g/mean":-0.00591278076171875,"train/train/tensor_act_model_layers_27_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/norm":6,"train/train/tensor_act_model_layers_21_self_attn_k_proj/max_abs":5.875,"train/train/layer_model_layers_43/grad/max_abs":0.00118255615234375,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_60_input_layernorm/norm":5792.608642580033,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/max_abs":0.000110626220703125,"train/train/layer_model_layers_67/grad/norm":0.058313653453812675,"train/train/tensor_param_model_layers_40_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_v_proj/max_abs":2.78125,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/max_abs":0.00118255615234375,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/mean":-1.237494871020317e-07,"train/train/tensor_act_model_layers_82_mlp_waleed_W_u/std":0.5937500207831982,"train/train/tensor_act_model_layers_81_mlp_waleed/std":0.31884883980619,"train/train/tensor_param_model_layers_6_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_73/grad/norm":0.060774616749859345,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/max_abs":0.0002193450927734375,"train/train/tensor_grad_model_layers_66_mlp_waleed_W_u_weight/mean":-2.9313378036022186e-07,"train/train/tensor_act_model_layers_12_self_attn_o_proj/max_abs":1.1015625,"train/train/tensor_act_model_layers_2_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_17/act/std":0.941668314215423,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/norm":2.796875,"train/train/tensor_act_model_layers_26_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/norm":0.006806822491135843,"train/train/layer__model_layers_85/param/max_abs":1,"train/train/tensor_param_model_layers_70_mlp_waleed_W_u_weight/std":0.033447265625,"train/train/tensor_act_model_layers_12_self_attn_o_proj/norm":566.6216797264751,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/mean":6.315531209111214e-08,"train/train/tensor_act_model_layers_68_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn/max_abs":0.94140625,"train/train/tensor_act_model_layers_40/max_abs":25,"train/train/tensor_grad_model_layers_69_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68/max_abs":25,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/norm":0.02116406694414297,"train/train/tensor_grad_model_layers_79_mlp_waleed_W_g_weight/std":7.141266319044166e-05,"train/train/tensor_param_model_layers_18_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_post_attention_layernorm/std":1.0000011260174713,"train/train/tensor_act_model_layers_42_mlp_waleed/max_abs":3.15625,"train/train/tensor_grad_model_layers_42_mlp_waleed_W_u_weight/norm":0.01309421050051735,"train/train/tensor_act_model_layers_6_mlp_waleed/max_abs":6.5625,"train/train/tensor_act_model_layers_31_self_attn_q_proj/max_abs":5.84375,"train/train/tensor_act_model_layers_55_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/norm":0.026229427882240353,"train/train/tensor_act_model_layers_57_input_layernorm/std":1.0000009156423744,"train/train/tensor_act_model_layers_2/std":3.406268076940295,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/std":0.0419921875,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/norm":0.01387107015708178,"train/train/tensor_act_model_layers_84_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/std":2.24063612705113e-05,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/max_abs":0.2373046875,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn_q_proj/mean":-0.018341064453125,"train/train/tensor_act_model_layers_41_mlp_waleed_W_u/norm":2509.881268707848,"train/train/tensor_act_model_layers_2_mlp/mean":-0.00658416748046875,"train/train/tensor_act_model_layers_11_self_attn_o_proj/max_abs":1.2890625,"train/train/tensor_act_model_layers_41_self_attn_v_proj/max_abs":1.890625,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/max_abs":9.965896606445312e-05,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/std":3.779045569968064e-05,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/max_abs":0.00015544891357421875,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_q_proj/mean":-0.033935546875,"train/train/tensor_act_model_layers_82_mlp_down_proj/max_abs":4.1875,"train/train/tensor_act_model_layers_20_self_attn_o_proj/std":0.10132224757214221,"train/train/tensor_act_model_layers_43_mlp_down_proj/norm":387.8941567164412,"train/train/tensor_act_model_layers_83_self_attn_k_proj/std":1.1621145745169101,"train/train/tensor_act_model_layers_74_mlp_waleed_W_u/max_abs":3.84375,"train/train/tensor_act_model_layers_22_post_attention_layernorm/norm":5792.344360351701,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/std":0.0245361328125,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/max_abs":0.0006103515625,"train/train/tensor_act_model_layers_61_mlp_down_proj/max_abs":0.7890625,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_input_layernorm/max_abs":4.625,"train/train/tensor_param_model_layers_50_mlp_waleed_W_g_weight/max_abs":0.138671875,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/norm":5.5,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/norm":0.0016188435420004573,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/max_abs":0.000354766845703125,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/mean":4.132380126975477e-08,"train/train/tensor_act_model_layers_43_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_waleed_W_g_weight/std":0.0306396484375,"train/train/layer_model_layers_65/act/max_abs":24.5,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp_waleed_W_u/std":0.6494164739291515,"train/train/tensor_act_model_layers_18_mlp_waleed_W_g/norm":2125.6791557336064,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/std":0.059326171875,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/max_abs":0.1796875,"train/train/tensor_param_model_layers_49_mlp_waleed_W_g_weight/norm":4.875,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/mean":4.8160552978515625e-05,"train/train/tensor_act_model_layers_10_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/max_abs":0.0001659393310546875,"train/train/tensor_grad_model_layers_4_mlp_waleed_W_g_weight/mean":7.506459951400757e-07,"train/train/layer_model_layers_22/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_o_proj/max_abs":0.50390625,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_89/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/std":0.02783203125,"train/train/tensor_param_model_layers_92_mlp_waleed_W_u_weight/mean":0.00023555755615234375,"train/train/tensor_act_model_layers_24_mlp_down_proj/mean":0.0014057159423828125,"train/train/tensor_act_model_layers_40_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_mlp_waleed_W_u_weight/std":0.0250244140625,"train/train/tensor_act_model_layers_62_mlp/mean":0.00017702579498291016,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/mean":-9.441375732421875e-05,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/max_abs":0.00020885467529296875,"train/train/tensor_act_model_layers_71_post_attention_layernorm/norm":5792.604125976885,"train/train/tensor_act_model_layers_5_self_attn_v_proj/mean":-0.0234375,"train/train/tensor_param_model_layers_45_mlp_waleed_W_g_weight/mean":2.905726432800293e-06,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/norm":0.0232559163554288,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/norm":0.001561204179234759,"train/train/tensor_act_model_layers_84_self_attn_q_proj/std":1.5918004743851297,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_mlp_waleed_W_u_weight/std":0.060302734375,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/norm":0.03847868472990067,"train/train/tensor_grad_model_layers_34_mlp_waleed_W_g_weight/mean":-8.119968697428703e-08,"train/train/layer__model_layers_48/param/std":0.049342161756337426,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/mean":-1.1606607586145401e-07,"train/train/tensor_act_model_layers_12_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_o_proj/max_abs":0.8984375,"train/train/tensor_act_model_layers_60/max_abs":24.125,"train/train/tensor_act_model_layers_22_mlp_waleed/norm":872.4542839620597,"train/train/tensor_act_model_layers_91_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/norm":2.875,"train/train/tensor_act_model_layers_65_mlp/mean":0.0016021728515625,"train/train/tensor_act_model_layers_53_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_9/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/max_abs":0.1044921875,"train/train/tensor_act_model_layers_7_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/mean":-0.000209808349609375,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/max_abs":0.232421875,"train/train/tensor_act_model_layers_86_mlp/norm":2591.82479131875,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/max_abs":0.263671875,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_24_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_v_proj/norm":1938.3808239112666,"train/train/tensor_param_model_layers_12_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/std":3.055101738626429e-05,"train/train/tensor_act_model_layers_67_self_attn_k_proj/norm":6900.583700364486,"train/train/tensor_act_model_layers_46_self_attn/std":0.0885019313312692,"train/train/tensor_param_model_layers_83_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/max_abs":0.166015625,"train/train/layer_model_layers_48/grad/std":6.672176253106464e-05,"train/train/tensor_act_model_layers_51_input_layernorm/norm":5792.605834964955,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/mean":-1.0564923286437988e-05,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/max_abs":0.142578125,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/std":1.5104138735300174e-05,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/std":0.043212890625,"train/train/tensor_act_model_layers_38_mlp_waleed_W_g/norm":2333.3181383985366,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_input_layernorm/norm":5792.611816407145,"train/train/tensor_act_model_layers_90_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_mlp_waleed_W_u_weight/max_abs":0.000640869140625,"train/train/tensor_act_model_layers_73_self_attn_v_proj/norm":2938.0318104012963,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/std":3.7441297664616245e-05,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/norm":0.0007521099153147035,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/max_abs":0.1279296875,"train/train/tensor_grad_model_layers_81_mlp_waleed_W_u_weight/std":8.119323858789421e-05,"train/train/tensor_act_model_layers_52_mlp_waleed_W_u/norm":2811.553574020338,"train/train/tensor_act_model_layers_29_self_attn_k_proj/max_abs":5.03125,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/max_abs":0.00037384033203125,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_59_input_layernorm/norm":5792.608276367994,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/mean":7.666647434234619e-06,"train/train/tensor_param_model_layers_20_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_q_proj/std":1.455082209472556,"train/train/tensor_param_model_layers_26_mlp_waleed_W_g_weight/std":0.02392578125,"train/train/tensor_act_model_layers_13_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_6_self_attn_q_proj/max_abs":7.84375,"train/train/tensor_act_model_layers_51_self_attn_q_proj/max_abs":6.8125,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/mean":5.209585651755333e-08,"train/train/tensor_act_model_layers_50_self_attn_q_proj/mean":0.032257080078125,"train/train/layer__model_layers_33/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_o_proj/max_abs":0.349609375,"train/train/tensor_act_model_layers_82_self_attn_q_proj/norm":6937.741836466057,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/std":0.0235595703125,"train/train/layer_model_layers_32/act/norm":19129.811243330725,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/std":7.341915496090035e-05,"train/train/layer_model_layers_87/grad/mean":1.4742429538189713e-08,"train/train/tensor_grad_model_layers_50_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/std":0.0286865234375,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_6_mlp_waleed_W_u_weight/norm":4.1875,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/mean":-5.221366882324219e-05,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_14/param/std":0.047177952622093774,"train/train/layer_model_layers_72/act/max_abs":25.5,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/std":0.00014458629300664105,"train/train/tensor_act_model_layers_14_post_attention_layernorm/std":1.000000012892997,"train/train/tensor_act_model_layers_18_mlp_waleed_W_u/max_abs":2.09375,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/norm":6.6875,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/max_abs":0.00029754638671875,"train/train/global/act/max_abs":46.5,"train/train/tensor_act_model_layers_93/mean":-0.0604248046875,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/max_abs":0.30859375,"train/train/tensor_grad_model_layers_44_mlp_waleed_W_g_weight/norm":0.01406996052252512,"train/train/tensor_act_model_layers_53_mlp_waleed_W_g/mean":0.0071563720703125,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/norm":0.0008712408622883105,"train/train/layer__model_layers_28/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/max_abs":0.232421875,"train/train/tensor_param_model_layers_4_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/std":4.059090522588129e-05,"train/train/layer__model_layers_90/param/norm":26.789708407567634,"train/train/tensor_act_model_layers_59_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_85/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/norm":0.005084223020259823,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/norm":0.01361145565538659,"train/train/tensor_act_model_layers_15_input_layernorm/max_abs":5.125,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/mean":-3.114109858870506e-08,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/mean":1,"train/train/layer__model_layers_86/param/std":0.06126596907146449,"train/train/tensor_act_model_layers_74_self_attn/max_abs":3.96875,"train/train/tensor_param_model_layers_90_mlp_waleed_W_u_weight/std":0.053955078125,"train/train/tensor_act_model_layers_22/max_abs":26.375,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/norm":0.01852625626153842,"train/train/tensor_act_model_layers_93_self_attn_q_proj/std":1.234375217292863,"train/train/tensor_grad_model_layers_42_mlp_waleed_W_u_weight/std":3.612848307963307e-05,"train/train/tensor_act_model_rotary_emb/max_abs":1,"train/train/layer_model_layers_65/grad/frac_near_dtype_limit":0,"train/train/layer__model_layers_12/param/std":0.047560428922586745,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/mean":-9.931623935699463e-06,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/max_abs":0.00144195556640625,"train/train/tensor_act_model_layers_7_mlp_waleed_W_g/norm":2706.7548521262925,"train/train/tensor_param_model_layers_64_mlp_waleed_W_u_weight/norm":5.6875,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/norm":0.011161718153084104,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_9/param/norm":19.257379875972223,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_mlp_waleed/max_abs":3.3125,"train/train/tensor_act_model_layers_81_self_attn_v_proj/norm":3534.372717240547,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_mlp_waleed_W_u_weight/norm":5.1875,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_83_mlp_waleed_W_g_weight/std":0.041015625,"train/train/layer_model_layers_63/grad/std":5.465226787850507e-05,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_mlp_waleed_W_u/norm":6107.397608153,"train/train/tensor_act_lm_head/norm":88430.9205860547,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_mlp_down_proj/max_abs":0.89453125,"train/train/layer__model_layers_75/param/max_abs":1,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/mean":5.234032869338989e-07,"train/train/layer_model_layers_83/grad/mean":1.4839593652351041e-07,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn/max_abs":1.4375,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/norm":0.014161537366138885,"train/train/layer__model_layers_76/param/std":0.05593438814859213,"train/train/tensor_act_model_layers_18_self_attn_q_proj/max_abs":8.5,"train/train/tensor_param_model_layers_31_mlp_waleed_W_u_weight/mean":0.00010633468627929688,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/mean":-4.415051080286503e-08,"train/train/tensor_act_model_layers_57_mlp_down_proj/std":0.08496168655302186,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/max_abs":0.0002079010009765625,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_v_proj/mean":-0.011260986328125,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/std":3.905501961752415e-05,"train/train/layer__model_layers_57/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/max_abs":0.00118255615234375,"train/train/tensor_act_model_layers_49_input_layernorm/max_abs":5.25,"train/train/tensor_act_model_layers_66_mlp_down_proj/std":0.11010773532758321,"train/train/tensor_act_model_layers_1_self_attn_q_proj/mean":-0.03619384765625,"train/train/tensor_act_model_layers_39_post_attention_layernorm/norm":5792.6088867198105,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/mean":-0.000118255615234375,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/norm":5,"train/train/tensor_grad_model_layers_85_mlp_waleed_W_u_weight/mean":2.1443702280521393e-07,"train/train/tensor_param_model_layers_34_mlp_waleed_W_u_weight/mean":9.870529174804688e-05,"train/train/tensor_act_model_layers_88_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19/mean":-0.00760650634765625,"train/train/tensor_act_model_layers_57_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/max_abs":0.001007080078125,"train/train/tensor_act_model_layers_41_mlp/std":0.06542971819193676,"train/train/tensor_act_model_layers_27_mlp/mean":-0.002094268798828125,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/mean":1.230509951710701e-07,"train/train/tensor_act_model_layers_28_mlp_waleed/mean":0.0005197525024414062,"train/train/tensor_param_model_layers_62_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer_model_layers_38/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn_q_proj/mean":-0.02813720703125,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/norm":0.010137211855616243,"train/train/tensor_act_model_layers_25_mlp_waleed_W_g/mean":0.0009565353393554688,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/mean":-1.2514647096395493e-07,"train/train/tensor_act_model_layers_1_mlp_waleed/frac_near_user_limit":0,"train/train/layer__model_layers_36/param/std":0.04959071238978347,"train/train/tensor_act_model_layers_20_mlp_waleed_W_u/std":0.24316407360106562,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/mean":-9.822845458984375e-05,"train/train/tensor_param_model_layers_23_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_69/param/mean":0.0015438097687481718,"train/train/layer_model_layers_51/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_o_proj/std":0.23706407586112035,"train/train/tensor_act_model_layers_29_mlp_down_proj/max_abs":0.404296875,"train/train/tensor_param_model_layers_85_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_input_layernorm/max_abs":5.6875,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_62/param/mean":0.001586902160168438,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/mean":-3.6197889130562544e-09,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/norm":4.75,"train/train/tensor_act_model_layers_8_mlp_waleed_W_g/mean":-0.00319671630859375,"train/train/tensor_act_model_layers_67/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/layer__model_layers_21/param/std":0.048936325217419,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/mean":-0.000278472900390625,"train/train/tensor_param_model_layers_25_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_86_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/std":5.518369036180918e-05,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/mean":8.881092071533203e-06,"train/train/tensor_grad_model_layers_25_mlp_waleed_W_u_weight/mean":-2.7619535103440285e-08,"train/train/tensor_act_model_layers_46_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/max_abs":8.821487426757812e-05,"train/train/layer_model_layers_78/act/norm":22032.20697974082,"train/train/tensor_act_model_layers_23_self_attn/std":0.3632838940652383,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_20_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/max_abs":0.00017642974853515625,"train/train/tensor_act_model_layers_30_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_24/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp_waleed_W_u/mean":0.0040435791015625,"train/train/tensor_act_model_layers_44_self_attn_k_proj/norm":4914.677495569499,"train/train/tensor_act_model_layers_83_self_attn_q_proj/norm":7186.921844408905,"train/train/layer_model_layers_11/act/norm":22278.769822633883,"train/train/tensor_param_model_layers_22_mlp_waleed_W_g_weight/norm":4.375,"train/train/tensor_act_model_layers_24/std":2.7656473350246005,"train/train/tensor_grad_model_layers_86_mlp_waleed_W_g_weight/norm":0.032543817106367876,"train/train/tensor_act_model_layers_86_mlp_waleed_W_u/norm":5519.822457149062,"train/train/layer_model_layers_9/act/std":0.9709913102979806,"train/train/tensor_param_model_layers_70_mlp_waleed_W_u_weight/norm":6.0625,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/mean":4.291534423828125e-06,"train/train/tensor_act_model_layers_45_mlp_waleed_W_g/max_abs":2.703125,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/norm":0.013825709137311131,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/max_abs":0.32421875,"train/train/tensor_act_model_layers_66_mlp_waleed_W_u/norm":3290.250029251288,"train/train/tensor_act_model_layers_49_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/std":9.025061405342025e-05,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_mlp_waleed_W_u_weight/max_abs":0.000492095947265625,"train/train/tensor_param_model_layers_78_mlp_waleed_W_u_weight/max_abs":0.21484375,"train/train/tensor_act_model_layers_47/std":2.625025624140119,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/std":0.06396484375,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/norm":0.016128446205176575,"train/train/tensor_grad_model_layers_10_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_mlp_waleed_W_g_weight/mean":-1.9550323486328125e-05,"train/train/tensor_act_model_layers_84_mlp_down_proj/max_abs":4.34375,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/norm":5.75,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/norm":4.125,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/loss":8.985243988037109,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65/mean":0.0147705078125,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/mean":-5.953945219516754e-06,"train/train/layer_model_layers_29/act/norm":19848.68635227246,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/std":0.02587890625,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/max_abs":0.2265625,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/std":0.034912109375,"train/train/tensor_param_model_layers_60_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_input_layernorm/norm":5792.604492189733,"train/train/tensor_act_model_layers_22_mlp_down_proj/mean":0.0011615753173828125,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/norm":4.28125,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/norm":0.010955172974462716,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_down_proj/mean":-0.00231170654296875,"train/train/layer_model_layers_58/act/max_abs":23.875,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/norm":5.90625,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/norm":3.125,"train/train/tensor_act_model_layers_17_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/std":3.5299935652075e-05,"train/train/tensor_act_model_layers_44/std":2.6250252987125586,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_10/grad/mean":-7.613598370514869e-08,"train/train/tensor_param_model_layers_59_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/std":0.040283203125,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_83/grad/norm":0.0675596770149809,"train/train/tensor_act_model_layers_16_self_attn_o_proj/mean":-0.00014663022011518478,"train/train/tensor_act_model_layers_75_self_attn_o_proj/mean":0.0012979507446289062,"train/train/tensor_act_model_layers_34_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/std":4.531757249604833e-05,"train/train/tensor_act_model_layers_84_mlp_down_proj/std":0.48730588389156154,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp_waleed_W_g/mean":0.00298309326171875,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/mean":-0.00018787384033203125,"train/train/tensor_act_model_layers_10_self_attn_q_proj/norm":7705.299185081788,"train/train/tensor_param_model_layers_23_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/mean":7.271766662597656e-06,"train/train/tensor_act_model_layers_31_self_attn_q_proj/std":0.9697281159410587,"train/train/tensor_act_model_layers_15_mlp_down_proj/mean":0.0016002655029296875,"train/train/tensor_act_model_layers_28_mlp/mean":-0.00021696090698242188,"train/train/tensor_act_model_layers_80/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn_o_proj/mean":-0.0009546279907226562,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/mean":-3.4371623769402504e-08,"train/train/tensor_act_model_layers_26_mlp_waleed/norm":865.4921309157862,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/max_abs":0.000522613525390625,"train/train/tensor_act_model_layers_62_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn/max_abs":5.90625,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/norm":0.0005914072695456249,"train/train/tensor_param_model_layers_67_mlp_waleed_W_u_weight/mean":-0.000247955322265625,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/norm":0.015335713084168043,"train/train/tensor_act_model_layers_56_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_post_attention_layernorm/max_abs":4.6875,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_waleed_W_u_weight/max_abs":0.1533203125,"train/train/tensor_act_model_layers_80_mlp/std":0.281250030630163,"train/train/tensor_grad_model_layers_35_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_post_attention_layernorm/mean":0.004467010498046875,"train/train/tensor_act_model_layers_80_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/mean":3.296881914138794e-07,"train/train/layer_model_layers_37/act/std":0.840463180741408,"train/train/layer_model_layers_91/act/norm":32810.36543895693,"train/train/tensor_grad_model_layers_3_mlp_waleed_W_g_weight/std":0.00016362538685897748,"train/train/tensor_act_model_layers_61_mlp_waleed_W_g/max_abs":2.84375,"train/train/tensor_act_model_layers_90_input_layernorm/norm":5792.61499023545,"train/train/tensor_act_model_layers_74_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/max_abs":0.000713348388671875,"train/train/tensor_param_model_layers_88_mlp_waleed_W_u_weight/norm":8.875,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/norm":3.265625,"train/train/tensor_act_model_layers_53_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/std":4.058638159655312e-05,"train/train/tensor_act_model_layers_75_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/norm":0.004776634686142636,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/max_abs":0.1962890625,"train/train/tensor_act_model_layers_65_mlp_down_proj/norm":619.7475039864526,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/mean":2.9476359486579895e-07,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/norm":0.0014403248199177875,"train/train/layer_model_layers_55/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/std":7.620358141458487e-06,"train/train/tensor_param_model_layers_80_mlp_waleed_W_g_weight/mean":0.00021648406982421875,"train/train/tensor_act_model_layers_5/mean":-8.869171142578125e-05,"train/train/layer_model_layers_61/act/std":0.8610852512338474,"train/train/tensor_act_model_layers_86_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/layer_model_layers_24/act/max_abs":26.5,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/norm":2.953125,"train/train/tensor_act_model_layers_73_input_layernorm/mean":0.0008726119995117188,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/std":6.323537695779866e-05,"train/train/tensor_act_model_layers_88_mlp_waleed_W_g/mean":-0.018951416015625,"train/train/tensor_param_model_layers_12_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/norm":0.0027925668440599073,"train/train/tensor_act_model_layers_41_mlp/mean":0.00223541259765625,"train/train/tensor_act_model_layers_12_mlp_waleed_W_u/max_abs":2.359375,"train/train/tensor_param_model_layers_31_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/mean":0.00012063980102539062,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/max_abs":0.000743865966796875,"train/train/tensor_grad_model_layers_1_mlp_waleed_W_g_weight/mean":7.152557373046875e-07,"train/train/tensor_grad_model_layers_11_mlp_waleed_W_u_weight/std":5.299909512785648e-05,"eval/samples_per_second":558.347,"train/train/tensor_param_model_layers_4_mlp_waleed_W_u_weight/std":0.0234375,"train/train/layer__model_layers_92/param/max_abs":1,"train/train/layer_model_layers_32/grad/max_abs":0.000820159912109375,"train/train/tensor_act_model_layers_55_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/std":3.2739783251683895e-05,"train/train/tensor_grad_model_layers_61_mlp_waleed_W_u_weight/norm":0.018707350066442002,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_27/grad/mean":-5.4743896454991116e-08,"train/train/tensor_act_model_layers_46_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_waleed_W_g_weight/max_abs":0.000865936279296875,"train/train/tensor_param_model_layers_48_mlp_waleed_W_g_weight/norm":4.9375,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/mean":0.00023555755615234375,"train/train/tensor_act_model_layers_11_input_layernorm/mean":-0.00547027587890625,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_post_attention_layernorm/max_abs":5.03125,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/max_abs":0.00016021728515625,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/mean":-4.800967872142792e-07,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/mean":-7.2177499532699585e-09,"train/train/tensor_grad_model_layers_64_mlp_waleed_W_u_weight/std":4.853303707589598e-05,"train/train/tensor_act_model_layers_83_mlp_waleed/mean":-0.00021904706954956055,"train/train/tensor_grad_model_layers_5_mlp_waleed_W_g_weight/norm":0.031471954337876,"train/train/tensor_act_model_layers_64_post_attention_layernorm/std":1.0000012885421594,"train/train/tensor_act_model_layers_86_self_attn_v_proj/std":0.5390625073830934,"train/train/tensor_act_model_layers_44_mlp_down_proj/std":0.07299846879094012,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn/std":0.18408401210564038,"train/train/tensor_param_model_layers_17_mlp_waleed_W_g_weight/mean":-0.00010395050048828125,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/norm":0.026709070850629832,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/max_abs":0.2392578125,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/max_abs":0.2490234375,"train/train/tensor_act_model_layers_49_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/max_abs":0.0022430419921875,"train/train/tensor_act_model_layers_65_self_attn_o_proj/std":0.127686253762904,"train/train/tensor_act_model_layers_58_mlp_waleed/frac_near_user_limit":0,"train/train/layer_model_layers_48/act/std":0.8887163873164396,"train/train/tensor_param_model_layers_15_mlp_waleed_W_u_weight/norm":4.1875,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/norm":0.006066304260644397,"train/train/tensor_param_model_layers_72_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_17_mlp_waleed_W_g/norm":2168.2415666125867,"train/train/tensor_act_model_layers_18_mlp_waleed/max_abs":2.625,"train/train/layer_model_layers_23/grad/norm":0.07505549770181938,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5/norm":19129.65031223114,"train/train/tensor_param_model_layers_85_mlp_waleed_W_u_weight/mean":-0.000316619873046875,"train/train/tensor_param_model_layers_65_mlp_waleed_W_g_weight/std":0.031494140625,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/mean":4.3248292058706284e-08,"train/train/tensor_grad_model_layers_35_mlp_waleed_W_g_weight/mean":1.7345882952213287e-08,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_input_layernorm/mean":0.00595855712890625,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/mean":1.8358230590820312e-05,"train/train/tensor_param_model_layers_76_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp_waleed/std":0.10656759957040784,"train/train/tensor_param_model_layers_40_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_53_self_attn/std":0.15582413819778,"train/train/tensor_act_model_layers_46_self_attn_k_proj/mean":-0.02301025390625,"train/train/tensor_param_model_layers_75_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_post_attention_layernorm/mean":0.0011501312255859375,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_30/act/std":0.854287298970799,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/max_abs":0.107421875,"train/train/tensor_act_model_layers_45_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/mean":4.076957702636719e-05,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_44_mlp_waleed_W_g/max_abs":2.9375,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/mean":-1.5739351511001587e-07,"train/train/tensor_act_model_layers_60_self_attn/norm":406.7039494663982,"train/train/tensor_act_model_layers_37_self_attn_k_proj/mean":0.028228759765625,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_22/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/mean":0.00012969970703125,"train/train/layer_model_layers_66/act/std":0.8640426395900439,"train/train/tensor_act_model_layers_42_self_attn_o_proj/std":0.12305319068326172,"train/train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/norm":5.5,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/norm":2.875,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/std":0.00012343206683172525,"train/train/tensor_param_model_layers_72_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_o_proj/norm":1417.8294786161323,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/max_abs":0.000225067138671875,"train/train/tensor_act_model_layers_63/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/std":8.475901362993302e-05,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/norm":0.01748793032500504,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp/max_abs":0.89453125,"train/train/layer_model_layers_93/grad/norm":0.09424554594233846,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_mlp_waleed_W_u_weight/max_abs":0.1708984375,"train/train/tensor_act_model_layers_60_mlp_down_proj/max_abs":0.62109375,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/max_abs":0.2333984375,"train/train/tensor_act_model_layers_91_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_o_proj/std":0.39844903955417277,"train/train/layer__model_layers_13/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_33/act/norm":19302.155694191806,"train/train/tensor_grad_model_layers_87_mlp_waleed_W_g_weight/norm":0.03297118914652551,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/mean":0.000263214111328125,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/max_abs":0.1328125,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/max_abs":0.255859375,"train/train/tensor_param_model_layers_66_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp_waleed/max_abs":3.875,"train/train/tensor_act_model_layers_62_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/mean":4.1484832763671875e-05,"train/train/layer_model_layers_54/grad/max_abs":0.00095367431640625,"train/train/tensor_act_model_layers_34_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_post_attention_layernorm/norm":5792.611938479211,"train/train/tensor_param_model_layers_81_mlp_waleed_W_u_weight/max_abs":0.2431640625,"train/train/layer_model_layers_32/grad/std":3.620405842615563e-05,"train/train/tensor_param_model_layers_48_mlp_waleed_W_g_weight/max_abs":0.1328125,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_mlp_waleed_W_u/norm":2357.381664469328,"train/train/tensor_act_model_layers_88_self_attn_v_proj/norm":3358.7559086892347,"train/train/tensor_act_model_layers_82_self_attn_v_proj/std":0.5869165414987862,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/max_abs":0.2255859375,"train/train/tensor_act_model_layers_70_post_attention_layernorm/std":1.000001027975867,"train/train/layer_model_layers_27/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn_q_proj/std":0.9199239828000553,"train/train/tensor_grad_model_layers_90_mlp_waleed_W_g_weight/std":0.00010280125286367194,"train/train/tensor_act_model_layers_33/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn/norm":2180.5780806122607,"train/train/tensor_act_model_layers_74_self_attn_q_proj/norm":6927.696058818162,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/mean":0.00012493133544921875,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/std":7.128758777447903e-05,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/norm":4.8125,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/std":7.44026465046055e-05,"train/train/tensor_act_model_layers_41_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_90/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/max_abs":0.15234375,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_55/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/std":3.312239549076621e-05,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/max_abs":0.00124359130859375,"train/train/tensor_act_model_layers_82_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/layer_model_layers_85/act/max_abs":30,"train/train/tensor_act_model_layers_92_self_attn_q_proj/norm":7131.543083190089,"train/train/tensor_act_model_layers_46_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_input_layernorm/max_abs":3.921875,"train/train/tensor_param_model_layers_10_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/norm":0.004591425156566566,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/max_abs":0.2255859375,"train/train/tensor_act_model_layers_65/norm":15268.822586686972,"train/train/tensor_grad_model_layers_62_mlp_waleed_W_u_weight/norm":0.01739683978277787,"train/train/tensor_grad_model_layers_63_mlp_waleed_W_g_weight/std":5.013455895774863e-05,"train/train/tensor_act_model_layers_30_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn_k_proj/max_abs":5.65625,"train/train/tensor_param_model_layers_8_mlp_waleed_W_u_weight/norm":4.15625,"train/train/tensor_act_model_layers_10_self_attn_q_proj/std":1.3281252917121118,"train/train/layer__model_layers_11/param/std":0.047872284638796274,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/mean":-2.4400651454925537e-07,"train/train/tensor_act_model_layers_62_self_attn/max_abs":1.265625,"train/train/tensor_act_model_layers_67_mlp_waleed/std":0.18188528591241723,"train/train/layer__model_layers_14/param/mean":0.0015241530681735082,"train/train/tensor_act_model_layers_35_self_attn_k_proj/norm":5620.817041464501,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/norm":0.00383318426855867,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/norm":3.734375,"train/train/tensor_act_model_layers_17_mlp_waleed_W_g/std":0.2641615335984733,"train/train/layer__model_layers_41/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp_waleed/norm":710.1829224924893,"train/train/tensor_act_model_layers_53_input_layernorm/mean":0.0008077621459960938,"train/train/layer_model_layers_19/grad/max_abs":0.00121307373046875,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/mean":3.509223461151123e-06,"train/train/layer_model_layers_30/grad/max_abs":0.0012359619140625,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/norm":5.28125,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/norm":0.011860585184886292,"train/train/tensor_param_model_layers_45_mlp_waleed_W_g_weight/std":0.0263671875,"train/train/tensor_act_model_layers_55_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/mean":-7.3802657425403595e-06,"train/train/tensor_param_model_layers_57_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/max_abs":0.00037384033203125,"train/train/tensor_grad_model_layers_91_mlp_waleed_W_g_weight/norm":0.045010851770175615,"train/train/layer_model_layers_80/grad/mean":2.2712284219618334e-07,"train/train/tensor_act_model_layers_50_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/std":4.650618647821253e-05,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/max_abs":0.00096893310546875,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/norm":0.00100533739221528,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/max_abs":0.00118255615234375,"train/train/tensor_grad_model_layers_73_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_mlp_waleed_W_u_weight/max_abs":0.000911712646484375,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/mean":4.202127456665039e-06,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/norm":0.005736026215653069,"train/train/tensor_act_model_layers_3_self_attn/std":0.03816298857885024,"train/train/layer_model_layers_51/grad/max_abs":0.00121307373046875,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/std":4.214129566357764e-05,"train/train/tensor_act_model_layers_22_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn/norm":738.3666472597946,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/norm":0.008868441808102624,"train/train/tensor_grad_model_layers_49_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/mean":-4.207715392112732e-06,"train/train/tensor_grad_model_layers_82_mlp_waleed_W_u_weight/mean":-1.6806734493002295e-07,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/mean":7.390975952148438e-05,"train/train/tensor_act_model_layers_71/std":2.734399258078107,"train/train/tensor_act_model_layers_56_self_attn_o_proj/max_abs":1.8203125,"train/train/tensor_act_model_layers_39_self_attn_v_proj/mean":0.00421905517578125,"train/train/layer_model_layers_73/act/max_abs":25.75,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/std":0.02587890625,"train/train/tensor_act_model_layers_31_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/std":0.06005859375,"train/train/tensor_act_model_layers_65_mlp/max_abs":0.984375,"train/train/tensor_act_model_layers_79_mlp_waleed/mean":-0.0018062591552734375,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/mean":1.0550953447818756e-05,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/mean":3.905966877937317e-06,"train/train/tensor_act_model_layers_12_mlp_waleed/mean":-0.002063751220703125,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/std":0.05224609375,"train/train/layer_model_layers_83/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn/norm":723.4390312814463,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/mean":-3.3993273973464966e-07,"train/train/layer_model_layers_6/act/frac_near_user_limit":0,"train/train/layer__model_layers_54/param/std":0.04969836447657876,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/std":2.082731813848312e-05,"train/train/layer__model_layers_17/param/norm":19.55693311366841,"train/train/tensor_act_model_layers_85_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/norm":5.3125,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/max_abs":0.1826171875,"train/train/layer__model_layers_88/param/mean":0.001531508709078832,"train/train/tensor_act_model_layers_21_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/max_abs":0.2353515625,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_o_proj/norm":3496.356921138058,"train/train/tensor_param_model_layers_11_mlp_waleed_W_g_weight/std":0.0230712890625,"train/train/tensor_act_model_layers_58_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_post_attention_layernorm/std":1.0000011209625055,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_down_proj/norm":391.17678248616306,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/mean":-4.4459011405706406e-07,"train/train/tensor_param_model_layers_16_mlp_waleed_W_g_weight/norm":4.1875,"train/train/tensor_act_model_layers_5/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/std":0.04248046875,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_self_attn_q_proj/mean":-0.036865234375,"train/train/tensor_act_model_layers_44_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_mlp_waleed_W_g_weight/std":0.0289306640625,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/max_abs":0.1767578125,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_embed_tokens_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/std":1.0232840775104785e-05,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/max_abs":0.002410888671875,"train/train/tensor_act_model_layers_20_self_attn_v_proj/std":0.41455170875979225,"train/train/tensor_param_model_layers_13_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/norm":4.3125,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/norm":3.390625,"train/train/tensor_act_model_layers_36_post_attention_layernorm/mean":0.001995563507080078,"train/train/tensor_act_model_layers_74_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/std":2.3101657049863463e-05,"train/train/tensor_act_model_layers_67_self_attn_k_proj/mean":0.054931640625,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/norm":0.036807185062988,"train/train/tensor_act_model_layers_51_self_attn_v_proj/mean":0.003299713134765625,"train/train/tensor_grad_model_layers_61_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_waleed_W_g/std":0.7929690289379437,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/mean":-0.0002384185791015625,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/mean":0.0002498626708984375,"train/train/tensor_act_model_layers_1_self_attn_v_proj/mean":-0.0081329345703125,"train/train/layer_model_layers_79/grad/norm":0.06481484489184575,"train/train/tensor_act_model_layers_13_mlp_waleed_W_u/max_abs":3.046875,"train/train/tensor_act_model_layers_41_self_attn_q_proj/norm":4734.374904744579,"train/train/tensor_act_model_layers_68_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/norm":5.4375,"train/train/tensor_act_model_layers_55_mlp/std":0.08557225491694422,"train/train/tensor_act_model_layers_8_mlp_waleed/mean":0.00016880035400390625,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_5/act/norm":24706.508410580456,"train/train/tensor_grad_model_layers_75_mlp_waleed_W_g_weight/norm":0.023564629678582538,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/max_abs":0.138671875,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/std":3.487092343554096e-05,"train/train/layer__model_layers_50/param/mean":0.0015722503900156006,"train/train/tensor_act_model_layers_93_mlp_waleed_W_g/std":1.1484376567662873,"train/train/tensor_act_model_layers_2_self_attn/max_abs":0.90625,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_waleed_W_g/max_abs":1.7109375,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/std":0.0002928508858516676,"train/train/tensor_act_model_layers_10_self_attn/max_abs":1.28125,"train/train/tensor_param_model_layers_75_mlp_waleed_W_g_weight/mean":-0.0001621246337890625,"train/train/tensor_act_model_layers_17_self_attn/norm":1065.7751087551435,"train/train/layer__model_layers_22/param/frac_near_user_limit":0,"train/train/layer_model_layers_49/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/norm":0.015327755520146246,"train/train/tensor_act_model_layers_17_post_attention_layernorm/max_abs":5.21875,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_38/grad/norm":0.04189367846071716,"train/train/tensor_act_model_layers_24_mlp_down_proj/norm":486.1305667937912,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/mean":0.0002574920654296875,"train/train/tensor_act_model_layers_91_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_v_proj/mean":0.002933502197265625,"train/train/tensor_grad_model_layers_36_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp_waleed/std":0.11645533784851365,"train/train/tensor_act_model_layers_75_self_attn_v_proj/mean":0.0078582763671875,"train/train/tensor_act_model_layers_37_post_attention_layernorm/norm":5792.615966801428,"train/train/tensor_act_model_layers_14_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/mean":-2.561137080192566e-08,"train/train/tensor_act_model_layers_10_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_k_proj/std":0.9628926987074368,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/mean":3.427267074584961e-07,"train/train/tensor_act_model_layers_69/norm":15659.837660336325,"train/train/tensor_act_model_layers_63_post_attention_layernorm/max_abs":5.0625,"train/train/tensor_act_model_layers_92_post_attention_layernorm/mean":-0.004116058349609375,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/mean":-0.0002899169921875,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/std":0.0224609375,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/mean":1.9418075680732727e-07,"train/train/tensor_act_model_layers_87_mlp_waleed_W_g/mean":-0.0084991455078125,"train/train/tensor_act_model_layers_90_post_attention_layernorm/norm":5792.614257813046,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_8/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp_down_proj/std":0.07006878776098212,"train/train/tensor_act_model_layers_40_mlp/norm":389.3823485264705,"train/train/tensor_grad_model_layers_23_mlp_waleed_W_g_weight/norm":0.012742431935316993,"train/train/tensor_act_model_layers_6_self_attn_k_proj/max_abs":6.3125,"train/train/tensor_grad_model_layers_37_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_/norm":5.845167548307916,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_down_proj/mean":0.00174713134765625,"train/train/layer_model_layers_88/act/frac_near_user_limit":0,"train/train/layer_model_layers_62/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/max_abs":0.0002765655517578125,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/norm":3.75,"train/train/tensor_act_model_layers_60_self_attn/max_abs":1.2421875,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/max_abs":0.001373291015625,"train/train/layer_model_layers_10/act/mean":-0.004317641258239746,"train/train/tensor_act_model_layers_90_self_attn_q_proj/norm":7361.5192020651775,"train/train/tensor_act_model_layers_60_mlp_waleed/norm":1109.3774679599346,"train/train/tensor_act_model_layers_36_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/max_abs":0.16015625,"train/train/tensor_act_model_layers_84_post_attention_layernorm/mean":0.0157623291015625,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/mean":-2.919696271419525e-06,"train/train/tensor_act_model_layers_34_self_attn_v_proj/max_abs":2.28125,"train/train/tensor_act_model_layers_67_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/norm":5.09375,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/mean":1.1727679520845413e-06,"train/train/tensor_act_model_layers_28_self_attn/std":0.20092936609984285,"train/train/tensor_param_model_layers_77_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/norm":0.004130963128051746,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/max_abs":0.1923828125,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/std":3.182581430001361e-05,"train/train/tensor_act_model_layers_67_self_attn/mean":-0.0007352828979492188,"train/train/tensor_act_model_layers_88_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_mlp_waleed_W_g_weight/mean":-6.914138793945312e-05,"train/train/tensor_act_model_layers_23_mlp_waleed_W_u/std":0.23437500158324837,"train/train/tensor_act_model_layers_54_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_86/grad/std":8.572349507332663e-05,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/norm":0.008150764659688104,"train/train/tensor_grad_model_layers_3_mlp_waleed_W_g_weight/max_abs":0.003936767578125,"train/train/tensor_act_model_layers_57_self_attn_o_proj/norm":996.822521092103,"train/train/tensor_act_model_layers_69_self_attn/std":0.2209499399932727,"train/train/tensor_param_model_layers_70_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/std":0.022705078125,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp_down_proj/max_abs":1.2578125,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp/norm":330.01048833495656,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/std":6.193886422962778e-05,"train/train/tensor_act_model_layers_39_mlp_waleed_W_u/mean":-0.0082855224609375,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_mlp_waleed_W_u_weight/mean":4.514586180448532e-07,"train/train/tensor_act_model_layers_17/max_abs":26.125,"train/train/tensor_act_model_layers_37_mlp/norm":355.5914642876377,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/max_abs":0.00104522705078125,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/std":0.00011958673897913811,"train/train/layer__model_layers_45/param/norm":19.966457468167082,"train/train/tensor_act_model_layers_50_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_o_proj/std":0.08496171841038999,"train/train/tensor_act_model_layers_11_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/norm":0.030935673611608497,"train/train/tensor_grad_model_layers_22_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn/std":0.6035442013749587,"train/train/tensor_act_model_layers_54_post_attention_layernorm/std":1.0000010662390368,"train/train/tensor_act_model_layers_44_self_attn_k_proj/mean":-0.018951416015625,"train/train/layer_model_layers_71/act/norm":20295.870395283302,"train/train/tensor_act_model_layers_88_input_layernorm/mean":-0.003749847412109375,"train/train/tensor_grad_model_layers_47_mlp_waleed_W_u_weight/mean":1.0292569641023874e-07,"train/train/tensor_param_model_layers_68_mlp_waleed_W_u_weight/norm":5.875,"train/train/tensor_act_model_layers_36_mlp_waleed/std":0.10217307281868973,"train/train/layer__model_layers_52/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51/norm":15201.25668181474,"train/train/tensor_act_model_layers_92_mlp/std":1.3632870485461193,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/mean":1.490843715146184e-08,"train/train/tensor_act_model_layers_83_mlp_waleed_W_g/mean":0.03118896484375,"train/train/tensor_act_model_layers_66_self_attn_k_proj/std":0.8818376843170067,"train/train/tensor_grad_model_layers_19_mlp_waleed_W_g_weight/max_abs":0.00121307373046875,"train/train/tensor_act_model_layers_70_self_attn_o_proj/mean":0.00012445449829101562,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/mean":2.596527338027954e-06,"train/train/tensor_act_model_layers_86_input_layernorm/norm":5792.601562502953,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_q_proj/mean":-0.064208984375,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/std":4.621424321143798e-05,"train/train/tensor_act_model_layers_88_self_attn_q_proj/max_abs":6.34375,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/max_abs":0.208984375,"train/train/tensor_act_model_layers_55_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_embed_tokens/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/max_abs":0.2197265625,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/mean":-1.4798715710639954e-06,"train/train/tensor_act_model_layers_15/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3/std":3.3750182660147203,"train/train/tensor_act_model_layers_68_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_3/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_mlp_waleed_W_u_weight/mean":0.0002803802490234375,"train/train/layer__model_layers_35/param/norm":19.69658148678661,"train/train/tensor_grad_model_layers_66_mlp_waleed_W_u_weight/max_abs":0.000637054443359375,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/mean":5.690380930900574e-07,"train/train/tensor_act_model_layers_46_self_attn_k_proj/max_abs":4.5,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/max_abs":0.000270843505859375,"train/train/tensor_act_model_layers_50_mlp_down_proj/std":0.09350681639037162,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/std":0.05126953125,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/std":5.0542091531518115e-05,"train/train/tensor_param_model_layers_75_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_waleed_W_g_weight/max_abs":0.126953125,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/mean":0.0004329681396484375,"train/train/tensor_act_model_layers_92_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_q_proj/norm":7161.034324060716,"train/train/global/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn/mean":0.0011310577392578125,"train/train/tensor_act_model_layers_26_self_attn_k_proj/std":0.9238307118127589,"train/train/tensor_act_model_layers_28_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/norm":4.6875,"train/train/tensor_act_model_layers_61/std":2.5859749214970393,"train/train/tensor_act_model_layers_13/std":2.9648721630112562,"train/train/tensor_act_model_layers_87_mlp_down_proj/norm":3143.7238071516185,"train/train/tensor_param_model_layers_90_mlp_waleed_W_g_weight/norm":9.625,"train/train/tensor_act_model_layers_11_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_post_attention_layernorm/max_abs":5.5625,"train/train/layer__model_layers_3/param/std":0.046679419237439554,"train/train/tensor_act_model_layers_62_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_19/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/norm":0.004920764298227983,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/max_abs":0.255859375,"train/train/tensor_act_model_layers_82/mean":0.05023193359375,"train/train/tensor_act_model_layers_62_mlp_waleed_W_g/std":0.3847656322614795,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/norm":0.021455311624179227,"train/train/tensor_param_model_layers_74_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/norm":0.0009775473426874024,"train/train/tensor_param_model_layers_53_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_71_mlp_waleed_W_u/mean":-0.003444671630859375,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/norm":0.0055184313021644,"train/train/tensor_act_model_layers_13_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/max_abs":0.0011749267578125,"train/train/layer_model_layers_31/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/mean":-0.00017547607421875,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/max_abs":0.0005340576171875,"train/train/layer__model_layers_44/param/mean":0.0015703757727945837,"train/train/tensor_grad_model_layers_43_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/std":0.0230712890625,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/std":7.106489695690841e-05,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/max_abs":0.10205078125,"train/train/tensor_param_model_layers_28_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_65/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_mlp_waleed_W_u_weight/norm":0.015724757747118356,"train/train/tensor_act_model_layers_4_self_attn_v_proj/std":0.337890669445079,"train/train/layer__model_layers_71/param/mean":0.0015753449962217238,"train/train/tensor_param_model_layers_19_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_post_attention_layernorm/mean":0.0012216567993164062,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/max_abs":0.00101470947265625,"train/train/tensor_act_model_layers_55_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_self_attn/norm":902.5659256278218,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/norm":0.002963781654091532,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/max_abs":0.1904296875,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/norm":5.75,"train/train/tensor_act_model_layers_6_mlp_waleed_W_g/norm":2859.561300844709,"train/train/tensor_param_model_layers_61_mlp_waleed_W_g_weight/norm":5.40625,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/mean":7.890164852142334e-06,"train/train/tensor_act_model_layers_43_post_attention_layernorm/norm":5792.614135744847,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/norm":5.28125,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/std":1.628748442627688e-05,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/max_abs":0.00023651123046875,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/max_abs":0.10791015625,"train/train/tensor_param_model_layers_17_mlp_waleed_W_u_weight/max_abs":0.10546875,"train/train/layer_model_layers_67/grad/std":7.19817596952792e-05,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/std":0.041015625,"train/train/tensor_act_model_layers_73_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_self_attn_o_proj/std":0.056764184063560374,"train/train/tensor_act_model_layers_44_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_post_attention_layernorm/norm":5792.6146240248,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/std":8.755206861866366e-06,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_waleed_W_u_weight/mean":-6.67572021484375e-05,"train/train/tensor_act_model_layers_8_self_attn_v_proj/norm":3118.412160310561,"train/train/tensor_act_model_layers_35_self_attn_v_proj/std":0.3696299357315561,"train/train/tensor_act_model_layers_69_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/norm":3.359375,"train/train/tensor_act_model_layers_3_self_attn/mean":-0.0006875991821289062,"train/train/tensor_act_model_layers_26_self_attn_v_proj/max_abs":2.265625,"train/train/tensor_act_model_layers_79/max_abs":26.625,"train/train/tensor_act_model_layers_40_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_v_proj/max_abs":2.4375,"train/train/tensor_grad_model_layers_67_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_waleed_W_g_weight/max_abs":0.00066375732421875,"train/train/tensor_param_model_layers_53_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_input_layernorm/mean":-0.0055999755859375,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/std":0.0274658203125,"train/train/tensor_param_model_layers_32_mlp_waleed_W_u_weight/norm":4.5,"train/train/layer_model_layers_21/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/std":0.0267333984375,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/norm":0.018251450859821102,"train/train/tensor_act_model_layers_13_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/norm":0.021162306675081427,"train/train/tensor_act_model_layers_75_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_mlp_waleed_W_u_weight/mean":-3.3855438232421875e-05,"train/train/tensor_act_model_layers_42_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_waleed/norm":1267.9953344749,"train/train/tensor_act_model_layers_75_mlp/max_abs":1.453125,"train/train/tensor_act_model_layers_17_self_attn_k_proj/max_abs":5,"train/train/tensor_act_model_layers_48_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/mean":-3.341119736433029e-08,"train/train/tensor_grad_model_layers_2_mlp_waleed_W_g_weight/std":0.00020551130982916473,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/std":1.9331171509730177e-05,"train/train/tensor_act_model_layers_82_self_attn_o_proj/max_abs":4.0625,"train/train/tensor_act_model_layers_57_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/std":0.0002230279093191026,"train/train/tensor_act_model_layers_57/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn_o_proj/mean":0.00024271011352539062,"train/train/tensor_act_model_layers_75_mlp_waleed/std":0.23266642362406473,"train/train/tensor_act_model_layers_39_post_attention_layernorm/std":1.0000011447406492,"train/train/layer_model_layers_42/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/max_abs":0.00011444091796875,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp/mean":0.0002567768096923828,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/mean":1.8651189748197794e-07,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/mean":-2.193450927734375e-05,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/std":0.00012078249973269182,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/norm":0.0037782145214478834,"train/train/tensor_act_model_layers_55_self_attn_o_proj/std":0.06543248470708976,"train/train/tensor_act_model_layers_68_mlp/norm":661.322928585359,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/norm":0.0073095014845385815,"train/train/tensor_act_model_layers_25_mlp_waleed/norm":769.2274372326335,"train/train/tensor_act_model_layers_73_mlp_down_proj/norm":891.4172345806643,"train/train/tensor_act_model_layers_5_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72/max_abs":25.5,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/mean":5.930662155151367e-06,"train/train/tensor_param_model_layers_35_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77/max_abs":26,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/norm":3.3125,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_42/grad/norm":0.036083411887606234,"train/train/tensor_act_model_layers_70_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/max_abs":0.00040435791015625,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/norm":4.96875,"train/train/tensor_act_model_layers_76_mlp/norm":1085.3489554907305,"train/train/layer_model_layers_24/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn/norm":996.822521092103,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_k_proj/norm":5094.547413267713,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/max_abs":0.0008087158203125,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/norm":0.005997965827973181,"train/train/tensor_act_model_layers_66_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59/max_abs":24,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/std":0.0478515625,"train/train/tensor_act_model_layers_68_post_attention_layernorm/mean":0.005680084228515625,"train/train/tensor_act_model_layers_46_self_attn_k_proj/norm":5015.211761951276,"train/train/tensor_act_model_layers_58_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/norm":5.09375,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/std":0.0235595703125,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/mean":2.491474151611328e-05,"train/train/tensor_act_model_layers_53_self_attn_o_proj/max_abs":1.9375,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_v_proj/max_abs":3,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/mean":-4.26173210144043e-05,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/mean":1.239776611328125e-05,"train/train/tensor_act_model_layers_11_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/std":0.0255126953125,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/max_abs":0.158203125,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_mlp_waleed_W_u_weight/max_abs":0.18359375,"train/train/tensor_act_model_layers_51/mean":0.0062427520751953125,"train/train/tensor_act_model_layers_24_self_attn_q_proj/norm":6020.565692537399,"train/train/tensor_act_model_layers_14_mlp_waleed_W_u/norm":1908.8933860544507,"train/train/layer_model_layers_54/grad/std":4.142034298943368e-05,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/norm":0.0043704911823180485,"train/train/tensor_act_model_layers_30_self_attn/mean":0.00211334228515625,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/max_abs":0.00016498565673828125,"train/train/tensor_act_model_layers_21_mlp/std":0.07287628516813972,"train/train/tensor_act_model_layers_46_mlp_down_proj/mean":-0.0008440017700195312,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/std":8.955881235703471e-05,"train/train/tensor_param_model_layers_86_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/mean":-0.00010967254638671875,"train/train/tensor_grad_model_layers_7_mlp_waleed_W_g_weight/norm":0.03085802534025065,"train/train/layer_model_layers_87/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/norm":0.006481286284171338,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/mean":-2.501765266060829e-07,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/mean":-0.0001010894775390625,"train/train/tensor_act_model_layers_63_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/std":5.443608435338991e-05,"train/train/tensor_act_model_layers_71_self_attn_k_proj/max_abs":5.3125,"train/train/layer__model_layers_92/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_26/act/std":0.8612647872565848,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/std":0.0380859375,"train/train/tensor_act_model_layers_16_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/norm":8.1875,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/max_abs":0.00055694580078125,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/std":0.024169921875,"train/train/tensor_act_model_layers_79_self_attn_q_proj/max_abs":6.84375,"train/train/tensor_act_model_layers_17_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_21/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_mlp_waleed_W_g/std":0.28320314773069283,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/mean":-1.8905848264694214e-06,"train/train/tensor_param_model_layers_89_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_12_mlp_waleed/max_abs":3.171875,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/std":6.984080888924169e-05,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/max_abs":0.23828125,"train/train/tensor_act_model_layers_21_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_mlp_waleed/norm":849.2497471594872,"train/train/tensor_act_model_layers_13_mlp_waleed/max_abs":3.3125,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/norm":0.010226075687580471,"train/train/layer_model_layers_38/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_mlp_waleed_W_u_weight/norm":0.015764478184299897,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn/norm":470.45473175346694,"train/train/tensor_param_model_layers_32_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/mean":6.455229595303535e-08,"train/train/tensor_act_model_layers_51_mlp_down_proj/norm":469.63469681359896,"train/train/tensor_act_model_layers_39_input_layernorm/norm":5792.6079101564455,"train/train/tensor_grad_model_layers_89_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/norm":3.046875,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_42/grad/max_abs":0.001495361328125,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/norm":0.013827325722391516,"train/train/tensor_param_model_layers_40_mlp_waleed_W_u_weight/norm":4.75,"train/train/tensor_param_model_layers_88_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_62/grad/mean":-1.1204202861384186e-08,"train/train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_mlp_waleed_W_g_weight/mean":1.0609626770019531e-05,"train/train/tensor_act_model_layers_13/max_abs":26.125,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/std":0.00013194273206536004,"train/train/tensor_act_model_layers_89_mlp/norm":3399.882020861135,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/max_abs":0.00018978118896484375,"train/train/layer_model_layers_51/act/mean":0.006533175706863403,"train/train/tensor_act_model_layers_19_self_attn/mean":0.00018095970153808594,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_self_attn_k_proj/max_abs":4.5625,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/std":7.722380974476851e-05,"train/train/tensor_act_model_layers_20_mlp/max_abs":0.53515625,"train/train/tensor_act_model_layers_47_mlp_down_proj/max_abs":0.640625,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/std":0.00019859949220437992,"train/train/tensor_grad_model_layers_43_mlp_waleed_W_u_weight/std":4.0116656997374995e-05,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_mlp_waleed_W_u_weight/mean":-6.0558319091796875e-05,"train/train/tensor_grad_model_layers_4_mlp_waleed_W_u_weight/mean":5.19677996635437e-07,"train/train/tensor_act_model_layers_27_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_v_proj/std":0.40429710384433826,"train/train/layer__model_layers_15/param/norm":19.195520768609665,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/mean":3.440072759985924e-08,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/std":0.023681640625,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/mean":-2.9726652428507805e-07,"train/train/layer_model_layers_8/act/mean":0.005785584449768066,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/max_abs":0.095703125,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/mean":8.49468051455915e-09,"train/train/tensor_act_model_layers_35_self_attn_k_proj/mean":0.0014462471008300781,"train/train/layer__model_layers_82/param/std":0.058756078742044804,"train/train/tensor_act_model_layers_11_self_attn_v_proj/std":0.36865346518402814,"train/train/tensor_act_model_layers_56_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp/max_abs":0.4609375,"train/train/tensor_param_model_layers_63_mlp_waleed_W_g_weight/norm":5.5625,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_21/param/max_abs":1,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_post_attention_layernorm/std":1.0000002494689086,"train/train/tensor_act_model_layers_2_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/mean":-0.000446319580078125,"train/train/tensor_act_model_layers_65_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_input_layernorm/std":1.000000793435587,"train/train/tensor_act_model_layers_56_self_attn_o_proj/std":0.12672286889455264,"train/train/tensor_act_model_layers_1/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn/mean":-0.0002110600471496582,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/max_abs":0.00023651123046875,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/mean":0.00070953369140625,"train/train/tensor_act_model_layers_9_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/mean":-1.6421079635620117e-05,"train/train/tensor_grad_model_layers_72_mlp_waleed_W_u_weight/norm":0.021578754849533067,"train/train/tensor_act_model_layers_35_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51/max_abs":23.375,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/norm":0.008334931081350763,"train/train/tensor_param_model_layers_0_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_down_proj/mean":0.000247955322265625,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/std":0.03369140625,"train/train/tensor_act_model_layers_20_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp_waleed_W_u/max_abs":3.5625,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/std":0.00010469237337212012,"train/train/tensor_grad_model_layers_34_mlp_waleed_W_u_weight/std":3.743211443982284e-05,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/mean":-3.771856427192688e-07,"train/train/tensor_act_model_layers_18_self_attn_o_proj/std":0.10351739021326295,"train/train/tensor_act_model_layers_18_mlp_waleed_W_u/mean":-0.0025482177734375,"train/train/layer_model_layers_42/grad/std":4.454664839179478e-05,"train/train/tensor_grad_model_layers_59_mlp_waleed_W_g_weight/max_abs":0.0008697509765625,"train/train/tensor_act_model_layers_13_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_q_proj/norm":6845.470223951015,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_69/param/std":0.05314417279023677,"train/train/layer_model_layers_46/grad/max_abs":0.00092315673828125,"train/train/tensor_act_model_layers_75_self_attn/mean":0.0012979507446289062,"train/train/layer_model_layers_23/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_input_layernorm/std":1.0000013405433603,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/std":7.203327679347992e-05,"train/train/tensor_grad_model_layers_59_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_42/act/max_abs":24.75,"train/train/tensor_act_model_layers_4_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39/norm":15275.065699684332,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/std":3.4800661339107276e-05,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_down_proj/norm":749.5972536418375,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/max_abs":0.11572265625,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/mean":3.24249267578125e-05,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_9/grad/norm":0.05522375426570568,"train/train/tensor_param_model_layers_51_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp/mean":0.0016002655029296875,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_mlp_down_proj/norm":370.04833947114315,"train/train/tensor_param_model_layers_71_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_43/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_30_post_attention_layernorm/std":1.0000003446183972,"train/train/tensor_act_model_layers_48_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/norm":0.015476895798122822,"train/train/layer__model_layers_93/param/mean":0.001708984375,"train/train/tensor_act_model_layers_61_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27/mean":0.002463817596435547,"train/train/tensor_act_model_layers_88_mlp_waleed_W_u/max_abs":5.15625,"train/train/tensor_act_model_layers_64_self_attn/mean":-0.0014476776123046875,"train/train/tensor_act_model_layers_59_self_attn/std":0.17236677124180275,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/max_abs":0.18359375,"train/train/tensor_act_model_layers_45_input_layernorm/max_abs":5.125,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/max_abs":0.0002956390380859375,"train/train/tensor_act_model_layers_53_self_attn_o_proj/mean":0.00016305968165397644,"train/train/tensor_grad_model_layers_53_mlp_waleed_W_g_weight/mean":-1.047446858137846e-07,"train/train/tensor_act_model_layers_90_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn/max_abs":0.953125,"train/train/tensor_act_model_layers_31_self_attn_o_proj/mean":0.00020304322242736816,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/max_abs":0.001007080078125,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/max_abs":0.004241943359375,"train/train/tensor_act_model_layers_14_input_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_26/param/std":0.04826510623592567,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/mean":-0.0001583099365234375,"train/train/tensor_act_model_layers_15_mlp_waleed_W_g/max_abs":2.28125,"train/train/tensor_act_model_layers_32_input_layernorm/max_abs":5.4375,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/norm":0.013552072696060983,"train/train/tensor_param_model_layers_53_mlp_waleed_W_g_weight/norm":5.125,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/norm":5.6875,"train/train/tensor_act_model_layers_6/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/max_abs":0.00106048583984375,"train/train/tensor_act_model_layers_63_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_v_proj/mean":-0.023162841796875,"train/train/tensor_grad_model_layers_84_mlp_waleed_W_u_weight/std":8.69195651556432e-05,"train/train/layer__model_layers_25/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_k_proj/norm":6014.525656904828,"train/train/tensor_act_model_layers_1_mlp_waleed_W_u/mean":-0.018157958984375,"train/train/tensor_act_model_layers_80_mlp_waleed_W_u/std":0.566406587691042,"train/train/layer__model_layers_30/param/max_abs":1,"train/train/tensor_act_model_layers_32_self_attn_o_proj/norm":144.58706390062105,"train/train/tensor_act_model_layers_34_mlp/mean":0.0013427734375,"train/train/tensor_param_model_layers_80_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_85/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn/norm":599.6632659444573,"train/train/tensor_act_model_layers_41_self_attn_k_proj/max_abs":3.75,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/std":0.0263671875,"train/train/tensor_act_model_layers_41_self_attn_o_proj/std":0.04614261692701119,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp_waleed/mean":0.019287109375,"train/train/layer_model_layers_11/act/max_abs":26.25,"train/train/tensor_grad_model_layers_41_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/mean":-1.0820804163813591e-07,"train/train/tensor_act_model_layers_52_input_layernorm/std":1.0000013966894608,"train/train/tensor_act_model_layers_87_post_attention_layernorm/norm":5792.619873050183,"train/train/layer__model_layers_56/param/max_abs":1,"train/train/tensor_act_model_layers_17_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp_waleed/std":0.09277408160045669,"train/train/tensor_param_model_layers_88_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/std":8.229901722051664e-05,"train/train/tensor_grad_model_layers_51_mlp_waleed_W_u_weight/mean":1.0844087228178978e-07,"train/train/tensor_act_model_layers_73_mlp_waleed_W_g/mean":0.00408935546875,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/mean":1.546577550470829e-07,"train/train/tensor_act_model_layers_19_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_mlp_waleed_W_g/norm":2870.0925315304353,"train/train/layer_model_layers_24/act/mean":0.007695436477661133,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/max_abs":0.000507354736328125,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/norm":0.0007516744564832664,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/mean":-0.000225067138671875,"train/train/tensor_act_model_layers_45_post_attention_layernorm/mean":0.00470733642578125,"train/train/tensor_param_model_layers_82_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/max_abs":0.000217437744140625,"train/train/tensor_act_model_layers_86_self_attn_q_proj/norm":6715.63229393464,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/norm":5.4375,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/mean":1.9837170839309692e-07,"train/train/tensor_act_model_layers_81_mlp_waleed_W_u/norm":4713.480777587859,"train/train/layer__model_layers_53/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/max_abs":0.2177734375,"train/train/layer__model_layers_84/param/norm":24.277147336795977,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/std":0.032958984375,"train/train/tensor_act_model_layers_72_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/layer_model_layers_73/grad/std":7.498372404364748e-05,"train/train/tensor_grad_model_layers_2_mlp_waleed_W_u_weight/max_abs":0.0040283203125,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/std":0.037353515625,"train/train/tensor_act_model_layers_65_mlp/std":0.1069336280430778,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/norm":4.90625,"train/train/tensor_param_model_layers_4_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/max_abs":0.18359375,"train/train/layer__model_layers_88/param/std":0.0629062983786223,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_92_mlp_waleed/std":0.9462948826593884,"train/train/tensor_act_model_layers_45_post_attention_layernorm/std":1.0000014645909876,"train/train/tensor_grad_model_layers_27_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8/mean":-0.00272369384765625,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/std":0.021728515625,"train/train/tensor_act_model_layers_66_input_layernorm/std":1.0000011569566267,"train/train/tensor_act_model_layers_49_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/std":0.03515625,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/std":6.29525559199308e-05,"train/train/tensor_act_model_layers_24_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/std":0.00014237643888426237,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_q_proj/std":0.901368784360618,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_mlp_waleed_W_g_weight/norm":0.023533544841384126,"train/train/tensor_act_model_layers_68_mlp_waleed_W_g/mean":-0.005462646484375,"train/train/layer_model_layers_35/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/std":0.0224609375,"train/train/tensor_act_model_layers_21_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13/norm":17165.20227006771,"train/train/tensor_act_model_layers_44/mean":0.00952911376953125,"train/train/tensor_act_model_layers_44/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/mean":1.3187527656555176e-05,"train/train/layer_model_layers_46/grad/mean":7.280588800933171e-08,"train/train/tensor_act_model_layers_24_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/max_abs":0.0002460479736328125,"train/train/tensor_act_model_layers_92_mlp_waleed_W_u/max_abs":5.65625,"train/train/tensor_act_model_layers_54_mlp/std":0.07714846846204856,"train/train/tensor_act_model_layers_10_mlp/mean":-0.0006723403930664062,"train/train/tensor_act_model_layers_90_mlp_down_proj/std":0.7558674169252445,"train/train/tensor_grad_model_layers_15_mlp_waleed_W_g_weight/mean":2.450542524456978e-08,"train/train/tensor_act_model_layers_23_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_mlp_waleed_W_u_weight/max_abs":0.001007080078125,"train/train/layer_model_layers_0/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/norm":4.9375,"train/train/tensor_act_model_layers_83_self_attn/max_abs":3.296875,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/mean":0.000392913818359375,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/mean":2.477318048477173e-07,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_28/grad/std":5.0526312343421056e-05,"train/train/layer__model_layers_14/param/max_abs":1,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_67/param/norm":22.00179768933666,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/std":4.17038776921603e-05,"train/train/tensor_act_model_layers_20_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43/mean":0.007991790771484375,"train/train/tensor_grad_model_layers_53_mlp_waleed_W_g_weight/std":4.256325842047501e-05,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp_waleed/max_abs":2.421875,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/std":3.2558335262027025e-05,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/max_abs":0.12255859375,"train/train/tensor_act_model_layers_46_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn_q_proj/norm":6031.470574963658,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/norm":4.875,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/norm":3.828125,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/norm":0.0021604796524501107,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/max_abs":0.1220703125,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/norm":0.022493173437845713,"train/train/tensor_param_model_layers_59_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_mlp_waleed_W_u_weight/mean":1.0686926543712616e-07,"train/train/tensor_act_model_layers_70_mlp_waleed/frac_near_user_limit":0,"train/train/layer_model_layers_67/grad/mean":-4.220300596701374e-08,"train/train/tensor_act_model_layers_64_self_attn_q_proj/norm":6057.882219672233,"train/train/tensor_act_model_layers_8_self_attn_k_proj/std":2.0937500240197826,"train/train/layer__model_layers_59/param/max_abs":1,"train/train/tensor_act_model_layers_9_self_attn_k_proj/std":1.3046876913059118,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_norm/max_abs":6.5625,"train/train/tensor_act_model_layers_40_mlp_down_proj/norm":389.3823485264705,"train/train/tensor_act_model_layers_41_mlp_waleed/std":0.10302734902896574,"train/train/tensor_act_model_layers_44_mlp_waleed_W_u/mean":-0.0138092041015625,"train/train/tensor_grad_model_layers_52_mlp_waleed_W_g_weight/max_abs":0.0004749298095703125,"train/train/tensor_act_model_layers_91_post_attention_layernorm/mean":-0.003589630126953125,"train/train/tensor_act_model_layers_41_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_mlp_waleed_W_g/std":0.946290621784518,"train/train/tensor_act_model_layers_54_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_o_proj/max_abs":1.2890625,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/max_abs":0.000362396240234375,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_40/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/norm":0.0007497699895214798,"train/train/tensor_act_model_layers_43_mlp_waleed_W_g/norm":2573.077077171482,"train/train/tensor_act_model_layers_47/norm":15195.883787872446,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/std":0.0003452301025390625,"train/train/tensor_param_model_layers_89_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/max_abs":0.2431640625,"train/train/tensor_act_model_layers_56_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/std":0.038330078125,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/mean":-6.437301635742188e-05,"train/train/tensor_act_model_layers_58_self_attn_o_proj/max_abs":2.578125,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/std":5.843217449374236e-06,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/max_abs":0.000881195068359375,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_54_self_attn_k_proj/norm":4716.426000995855,"train/train/layer__model_layers_4/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/std":0.026611328125,"train/train/tensor_act_model_layers_3_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_o_proj/norm":1682.9537174894388,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/std":3.993871797857776e-05,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/std":0.0361328125,"train/train/layer_model_layers_18/act/mean":-0.00040271878242492676,"train/train/layer__model_layers_33/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_v_proj/mean":-0.0011396408081054688,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/std":0.023681640625,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_mlp_waleed_W_g/mean":0.0011653900146484375,"train/train/tensor_act_model_layers_86_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_85/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_15_self_attn_o_proj/max_abs":1.046875,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/max_abs":0.12890625,"train/train/tensor_act_model_layers_54_mlp_waleed_W_u/norm":2826.477621022336,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_62/param/norm":20.936718735423895,"train/train/tensor_param_model_layers_30_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_mlp_waleed_W_g/norm":3151.466295318943,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/std":9.561380191273135e-06,"train/train/layer__model_layers_54/param/norm":20.146363854614062,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_mlp_waleed_W_u_weight/mean":-2.6333145797252655e-07,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/norm":5.4375,"train/train/tensor_act_model_layers_18_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33/norm":15472.532579501742,"train/train/tensor_act_model_layers_69_self_attn_q_proj/mean":0.0885009765625,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/max_abs":0.000553131103515625,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/mean":-5.0640664994716644e-09,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/max_abs":0.000152587890625,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn_q_proj/std":0.7929687921779008,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_down_proj/std":0.13623138613970398,"train/train/tensor_act_model_layers_12_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer_model_layers_85/act/mean":0.024183273315429688,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/max_abs":0.0009918212890625,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/max_abs":0.0002613067626953125,"train/train/layer_model_layers_0/grad/norm":2.6019221564149277,"train/train/tensor_act_model_layers_82_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn/norm":1431.7324613888693,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/std":0.0252685546875,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/max_abs":0.24609375,"train/train/tensor_act_model_layers_26_self_attn_q_proj/mean":0.04644775390625,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_mlp_waleed_W_g_weight/norm":0.03000366854092379,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/norm":0.014282422186661066,"train/train/tensor_act_model_layers_47_mlp_waleed_W_u/norm":2523.754759044816,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/std":0.036376953125,"train/train/tensor_act_model_layers_64_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_down_proj/max_abs":1.8203125,"train/train/layer_model_layers_89/act/norm":27872.191767077646,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/norm":0.004424233196988736,"train/train/layer_model_layers_63/act/std":0.8542535220645063,"train/train/tensor_param_model_layers_14_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_v_proj/max_abs":3.890625,"train/train/tensor_grad_model_layers_80_mlp_waleed_W_u_weight/norm":0.027562425051044455,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_52_self_attn_o_proj/mean":-0.0006847381591796875,"train/train/tensor_grad_model_layers_4_mlp_waleed_W_g_weight/max_abs":0.0029754638671875,"train/train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_input_layernorm/std":1.0000010459547397,"train/train/tensor_act_model_layers_20_mlp_down_proj/std":0.05694590028195003,"train/train/layer__model_layers_19/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn_k_proj/mean":0.0078277587890625,"train/train/tensor_act_model_layers_65_self_attn_k_proj/max_abs":5.15625,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/max_abs":0.000583648681640625,"train/train/tensor_act_model_layers_64_post_attention_layernorm/norm":5792.609130861891,"train/train/tensor_act_model_layers_1_mlp/norm":13242.716038913999,"train/train/tensor_act_model_layers_59_self_attn_v_proj/mean":0.00557708740234375,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/norm":0.020665876600971207,"train/train/layer_model_layers_55/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_63/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_o_proj/mean":4.034209996461868e-05,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/mean":8.026836439967155e-08,"train/train/tensor_act_model_layers_48_self_attn_v_proj/norm":3023.831337406949,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/mean":-1.1187512427568436e-07,"train/train/tensor_act_model_layers_42_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_post_attention_layernorm/mean":0.001323699951171875,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/std":4.25015896962851e-05,"train/train/tensor_act_model_layers_72_self_attn/norm":1658.9633491716186,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/mean":-5.762558430433273e-09,"train/train/tensor_act_model_layers_22_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_mlp_waleed_W_g_weight/max_abs":0.1650390625,"train/train/tensor_act_model_layers_88_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/norm":5.09375,"train/train/tensor_act_model_layers_58_self_attn_q_proj/std":1.123052341613733,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/std":0.022216796875,"train/train/layer_model_layers_76/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/max_abs":0.158203125,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_waleed_W_u_weight/norm":5.34375,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/mean":-0.00015926361083984375,"train/train/tensor_act_model_layers_31_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp/norm":1095.5890614664113,"train/train/layer_model_layers_90/grad/norm":0.08421947043682237,"train/train/tensor_param_model_layers_54_mlp_waleed_W_g_weight/max_abs":0.1474609375,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/mean":9.584426879882812e-05,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/max_abs":0.2421875,"train/train/tensor_act_model_layers_44_self_attn/max_abs":0.875,"train/train/tensor_act_model_layers_77/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_mlp_waleed_W_g/std":0.2485356749773782,"train/train/tensor_act_model_layers_64_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn_o_proj/mean":0.0013713836669921875,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/norm":0.015397592555243424,"train/train/tensor_act_model_layers_0_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/norm":0.01309819288147093,"train/train/tensor_act_model_layers_35_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/max_abs":0.00048828125,"train/train/tensor_act_model_layers_4_mlp_down_proj/std":0.24560693622856009,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp/mean":-0.0006818771362304688,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/max_abs":0.00018405914306640625,"train/train/tensor_act_model_layers_93_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn_k_proj/mean":-0.0704345703125,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_2_post_attention_layernorm/max_abs":4.3125,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/norm":0.0023811779122066582,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/std":0.0390625,"train/train/tensor_act_model_layers_76_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_mlp_waleed_W_g_weight/max_abs":0.1591796875,"train/train/tensor_param_model_layers_7_mlp_waleed_W_u_weight/norm":4.21875,"train/train/tensor_grad_model_layers_37_mlp_waleed_W_u_weight/norm":0.014497166623970926,"train/train/tensor_act_model_layers_68_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_input_layernorm/norm":5792.604003908257,"train/train/tensor_grad_model_layers_48_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/mean":-0.0002307891845703125,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/max_abs":0.2314453125,"train/train/tensor_act_model_layers_69_mlp/norm":724.7051076307085,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/mean":0.0001277923583984375,"train/train/tensor_act_model_layers_61_self_attn_o_proj/std":0.12475659434384297,"train/train/tensor_act_model_layers_8_self_attn_o_proj/max_abs":1.9453125,"train/train/tensor_act_model_layers_4_mlp_waleed_W_g/max_abs":2.65625,"train/train/tensor_param_model_layers_13_mlp_waleed_W_g_weight/norm":4.21875,"train/train/tensor_param_model_layers_18_mlp_waleed_W_g_weight/mean":-0.00019168853759765625,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_waleed_W_g/norm":6494.135270349621,"train/train/tensor_act_model_layers_71_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/norm":0.000553965760939622,"train/train/layer__model_layers_50/param/norm":20.20433339386937,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/norm":5.40625,"train/train/tensor_act_model_layers_58_self_attn/norm":1144.703296227091,"train/train/tensor_act_model_layers_3_self_attn_q_proj/max_abs":4.46875,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_68_self_attn_o_proj/std":0.14062502625165263,"train/train/tensor_act_model_layers_82_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_q_proj/max_abs":5.625,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/std":0.00011363666206926816,"train/train/tensor_grad_model_layers_77_mlp_waleed_W_g_weight/std":6.461719777775612e-05,"train/train/tensor_act_model_layers_67_mlp_down_proj/std":0.12939547884792055,"train/train/tensor_act_model_layers_49_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/mean":1.6792910173535347e-08,"train/train/tensor_act_model_layers_70_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_k_proj/norm":5265.673877819053,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/std":1.2972881172127056e-05,"train/train/tensor_grad_model_layers_47_mlp_waleed_W_g_weight/max_abs":0.00046539306640625,"train/train/tensor_act_model_layers_39_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_mlp_waleed_W_u_weight/mean":5.2299583330750465e-08,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/norm":0.003000787420547048,"train/train/tensor_act_model_layers_88/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/std":0.0260009765625,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/mean":-4.377216100692749e-07,"train/train/tensor_param_model_layers_19_mlp_waleed_W_g_weight/mean":0.00012063980102539062,"train/train/tensor_param_model_layers_78_mlp_waleed_W_u_weight/norm":6.84375,"train/train/tensor_act_model_layers_9_self_attn_q_proj/std":1.197270458703295,"train/train/tensor_act_model_layers_82_self_attn_k_proj/std":1.1523503966058979,"train/train/layer_model_layers_53/grad/norm":0.03487651611555315,"train/train/tensor_act_model_layers_11_mlp_down_proj/max_abs":0.4609375,"train/train/tensor_act_model_layers_55_mlp_down_proj/mean":-0.00038242340087890625,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/norm":3.140625,"train/train/tensor_act_model_layers_8_self_attn_q_proj/std":1.8671876396594136,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/norm":0.021027740758230305,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/mean":-3.1315721571445465e-07,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/max_abs":0.00010204315185546875,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/norm":0.007615771528829377,"train/train/tensor_act_model_layers_77_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/std":0.04541015625,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/std":4.692639585341145e-05,"train/train/tensor_act_model_layers_54_input_layernorm/max_abs":5.03125,"train/train/tensor_act_model_layers_67_mlp_waleed_W_u/max_abs":2.890625,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/max_abs":0.0004863739013671875,"train/train/tensor_act_model_layers_9_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/max_abs":0.0018157958984375,"train/train/layer_model_layers_41/grad/max_abs":0.000850677490234375,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_k_proj/mean":0.080078125,"train/train/tensor_act_model_layers_79_self_attn_q_proj/std":1.1543019225240938,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/norm":3.484375,"train/train/tensor_grad_model_layers_18_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_mlp_waleed_W_u_weight/mean":0.0002536773681640625,"train/train/tensor_act_model_layers_0_mlp_down_proj/std":2.3281250358027896,"train/train/tensor_act_model_layers_38_mlp_down_proj/max_abs":0.546875,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48/std":2.664098667070611,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_o_proj/norm":1165.626200868957,"train/train/layer__model_layers_47/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_down_proj/std":0.16796876542096842,"train/train/tensor_act_model_layers_66_self_attn_k_proj/mean":0.05914306640625,"train/train/tensor_act_model_layers_17_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_20/act/norm":20439.144824781964,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/mean":-2.1219253540039062e-05,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_19/param/max_abs":1,"train/train/tensor_act_model_layers_55_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32/norm":15508.351272923499,"train/train/tensor_act_model_layers_41_mlp_down_proj/max_abs":0.478515625,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/max_abs":0.181640625,"train/train/tensor_param_model_layers_10_mlp_waleed_W_g_weight/mean":-2.0265579223632812e-05,"train/train/tensor_act_model_layers_57_mlp/std":0.08496168655302186,"train/train/tensor_act_model_layers_30_input_layernorm/std":1.0000003027757427,"train/train/tensor_act_model_layers_79_mlp_waleed_W_g/max_abs":3.109375,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/norm":0.02222333539386673,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/mean":0.00010967254638671875,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/max_abs":0.1171875,"train/train/tensor_act_model_layers_9_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_q_proj/mean":-0.0010099411010742188,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/norm":0.0408570995390126,"train/train/tensor_act_model_layers_10_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_23/mean":0.004241943359375,"train/train/layer_model_layers_1/act/max_abs":27.375,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/norm":4.71875,"train/train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/max_abs":0.0004520416259765625,"train/train/tensor_act_model_layers_88_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_mlp_down_proj/max_abs":1.484375,"train/train/tensor_act_model_layers_20_mlp_waleed_W_g/norm":2036.3166154725961,"train/train/tensor_act_model_layers_13_self_attn_q_proj/mean":-0.022003173828125,"train/train/tensor_param_model_layers_49_mlp_waleed_W_u_weight/max_abs":0.1376953125,"train/train/tensor_act_model_layers_57_self_attn_q_proj/max_abs":5.1875,"train/train/layer_model_layers_5/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp_waleed_W_u/std":0.2968750128794739,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_post_attention_layernorm/max_abs":5.5625,"train/train/tensor_param_model_layers_56_mlp_waleed_W_u_weight/mean":2.9325485229492188e-05,"train/train/tensor_act_model_layers_70_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_mlp_waleed_W_g/mean":0.002712249755859375,"train/train/tensor_param_model_layers_85_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/norm":0.01300084646385245,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/mean":4.7417415771633387e-08,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/norm":4.4375,"train/train/layer_model_layers_83/act/max_abs":28.875,"train/train/tensor_act_model_layers_56_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/mean":-1.0279472917318344e-07,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/mean":4.824995994567871e-05,"train/train/layer_model_layers_77/act/std":0.9692742520639529,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/max_abs":0.31640625,"train/train/tensor_param_model_layers_12_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_mlp_waleed_W_u_weight/max_abs":0.158203125,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp_waleed_W_u/norm":2166.140923197032,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/std":0.00014778067478356133,"train/train/tensor_act_model_layers_85_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn/std":0.3339844865408371,"train/train/tensor_act_model_layers_79/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/mean":-0.00010395050048828125,"train/train/tensor_act_model_layers_54_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/max_abs":0.126953125,"train/train/tensor_act_model_layers_5_mlp_waleed_W_u/max_abs":2.140625,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/mean":1.6726553440093994e-06,"train/train/tensor_act_model_layers_49/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_waleed_W_u/std":0.7460937618585156,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/norm":0.0007462560939856989,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_80/grad/std":9.221743326782719e-05,"train/train/tensor_param_model_layers_11_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_62/act/norm":19470.458191430975,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/std":0.0267333984375,"train/train/tensor_act_model_layers_25_self_attn_q_proj/std":1.0546875282570165,"train/train/tensor_act_model_layers_59_mlp_waleed_W_g/mean":-0.012420654296875,"train/train/tensor_act_model_layers_44_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_mlp_waleed_W_u_weight/std":0.0286865234375,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/std":3.844818001237537e-05,"train/train/tensor_grad_model_layers_83_mlp_waleed_W_u_weight/mean":1.2415694072842598e-07,"train/train/tensor_param_model_layers_73_input_layernorm_weight/std":0,"train/train/layer_model_layers_20/grad/mean":4.6733941144764705e-08,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/norm":0.0006752714562592509,"train/train/tensor_act_model_layers_10_post_attention_layernorm/max_abs":5.125,"train/train/layer_model_layers_61/act/mean":0.0031310487538576126,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/norm":6.125,"train/train/tensor_act_model_layers_18_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn/mean":0.0132904052734375,"train/train/tensor_act_model_layers_86_self_attn_k_proj/std":1.0371156304380917,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/std":5.867353561712097e-05,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/norm":3.046875,"train/train/tensor_act_model_layers_6_mlp_waleed_W_u/max_abs":2.859375,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/norm":0.0006449941305982294,"train/train/tensor_act_model_layers_24_mlp_waleed/mean":0.006072998046875,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/max_abs":0.00160980224609375,"train/train/tensor_grad_model_layers_30_mlp_waleed_W_g_weight/max_abs":0.0005035400390625,"train/train/layer__model_layers_80/param/mean":0.0014817926701443057,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_35/act/std":0.8535943394189763,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/norm":2.96875,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/max_abs":0.00010824203491210938,"train/train/tensor_act_model_layers_86_input_layernorm/std":1.000000160849582,"train/train/tensor_act_model_layers_55/mean":0.0056667327880859375,"train/train/tensor_act_model_layers_2_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn_o_proj/mean":-2.7298927307128906e-05,"train/train/tensor_act_model_layers_19_self_attn_o_proj/std":0.0629892414231707,"train/train/tensor_grad_model_layers_38_mlp_waleed_W_g_weight/norm":0.01292523016263537,"train/train/tensor_param_model_layers_72_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_k_proj/norm":5771.235696541307,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_mlp_waleed/mean":0.00179290771484375,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_mlp_waleed_W_u_weight/std":0.02392578125,"train/train/tensor_act_model_layers_67_mlp_waleed_W_g/norm":3535.4267890162096,"train/train/tensor_act_model_layers_91_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/std":0.0224609375,"train/train/tensor_act_model_layers_36_mlp/max_abs":0.54296875,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/std":0.0001518066774848348,"train/train/tensor_act_model_layers_47_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_waleed_W_g/max_abs":2.875,"train/train/layer_model_layers_3/grad/std":0.00017237037138829818,"train/train/tensor_act_model_layers_73_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_v_proj/std":0.4096688478287641,"train/train/tensor_act_model_layers_47_self_attn_v_proj/mean":-0.00017571449279785156,"train/train/tensor_param_model_layers_91_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_23_mlp/norm":307.68916356707376,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/std":2.614745132511968e-05,"train/train/tensor_act_model_layers_59_self_attn_o_proj/mean":0.0011310577392578125,"train/train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_17/param/mean":0.0015932699075541146,"train/train/tensor_grad_model_layers_49_mlp_waleed_W_u_weight/mean":9.132781997323036e-08,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_mlp_waleed_W_u_weight/std":4.889297187522442e-05,"train/train/tensor_act_model_layers_17_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn/mean":-0.0006694793701171875,"train/train/tensor_act_model_layers_52_self_attn_v_proj/mean":-0.00215911865234375,"train/train/layer_model_layers_47/grad/mean":1.3466401377259847e-08,"train/train/tensor_param_model_layers_63_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_mlp_waleed_W_u_weight/std":0.026123046875,"train/train/tensor_act_model_layers_77_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp/std":0.10205253385480936,"train/train/tensor_act_model_layers_70/std":2.718774310482802,"train/train/tensor_act_model_layers_28_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_mlp_waleed_W_g/std":0.350099197716658,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/mean":-0.00010013580322265625,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/max_abs":0.11865234375,"train/train/tensor_param_model_layers_3_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/mean":-9.5367431640625e-06,"train/train/tensor_act_model_layers_55_mlp_waleed_W_u/max_abs":2.828125,"train/train/tensor_param_model_layers_11_mlp_waleed_W_u_weight/std":0.02294921875,"train/train/tensor_act_model_layers_76_self_attn_k_proj/mean":0.00609588623046875,"train/train/tensor_act_model_layers_21_mlp_down_proj/max_abs":0.671875,"train/train/tensor_act_model_layers_75/norm":16510.4985199097,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/max_abs":0.00022125244140625,"train/train/tensor_act_model_layers_83_self_attn/norm":1935.1022345329607,"train/train/tensor_act_model_layers_5_mlp_waleed_W_g/max_abs":2.28125,"train/train/tensor_act_model_layers_64_mlp_waleed_W_u/std":0.3945312582813364,"train/train/tensor_act_model_layers_22_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_q_proj/std":0.9638686870961375,"train/train/tensor_grad_model_layers_15_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/max_abs":0.00098419189453125,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/mean":-2.595061232568696e-08,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_input_layernorm/max_abs":5.125,"train/train/tensor_act_model_layers_4_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp_down_proj/std":0.07080078248319954,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/std":0.0002157530659280177,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/norm":4.34375,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/max_abs":0.000362396240234375,"train/train/tensor_act_model_layers_58_mlp_waleed_W_u/std":0.3593753326844666,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/std":5.67268633713784e-05,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/max_abs":0.1826171875,"train/train/tensor_act_model_layers_75_mlp_waleed_W_u/max_abs":3.546875,"train/train/layer_model_layers_57/act/norm":19846.366147183315,"train/train/tensor_act_model_layers_2_mlp_waleed_W_g/std":0.4951182208812245,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/norm":6.0625,"train/train/tensor_grad_model_layers_16_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_38/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_waleed/mean":0.0022125244140625,"train/train/tensor_act_model_layers_35/norm":15404.053641932935,"train/train/tensor_act_model_layers_71_self_attn_q_proj/std":0.9882851758886472,"train/train/tensor_grad_model_layers_32_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_35/grad/mean":1.5860525150194926e-07,"train/train/tensor_act_model_layers_15_self_attn_q_proj/mean":0.004001617431640625,"train/train/layer_model_layers_43/act/std":0.8349551293269925,"train/train/tensor_act_model_layers_66_self_attn_v_proj/std":0.4843771021857665,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_mlp_waleed_W_u_weight/mean":4.863739013671875e-05,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/std":6.227340694753045e-05,"train/train/tensor_grad_model_layers_91_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_28_mlp_waleed_W_u/std":0.25976563060194025,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/norm":4.5,"train/train/tensor_act_model_layers_48_mlp_waleed/max_abs":4.96875,"train/train/tensor_act_model_layers_85_self_attn_q_proj/norm":7454.682374505141,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/std":0.0225830078125,"train/train/layer_model_layers_62/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/max_abs":0.0002117156982421875,"train/train/tensor_param_model_layers_84_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_waleed_W_u_weight/max_abs":0.00070953369140625,"train/train/tensor_grad_model_layers_40_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp_waleed_W_u/mean":-0.007965087890625,"train/train/tensor_act_model_layers_8_mlp_waleed_W_g/std":0.2773437578703315,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/std":0.024169921875,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/max_abs":0.000568389892578125,"train/train/tensor_act_model_layers_34_mlp/std":0.05969250884754629,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/max_abs":0.00057220458984375,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/mean":3.9637088775634766e-06,"train/train/tensor_act_model_layers_80_post_attention_layernorm/max_abs":5.28125,"train/train/tensor_param_model_layers_87_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp/std":0.07080078248319954,"train/train/tensor_act_model_layers_8_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_input_layernorm_weight/std":0.0003452301025390625,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/max_abs":0.10791015625,"train/train/tensor_act_model_layers_27_input_layernorm/mean":0.002185821533203125,"train/train/tensor_param_model_layers_26_mlp_waleed_W_u_weight/max_abs":0.111328125,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/mean":4.684552550315857e-07,"train/train/tensor_grad_model_layers_12_mlp_waleed_W_u_weight/norm":0.024181786909770887,"train/train/tensor_param_model_layers_17_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_41_mlp_waleed_W_g/std":0.3066407620526876,"train/train/tensor_act_model_layers_57_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp/mean":-0.00044536590576171875,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/mean":-5.7334545999765396e-08,"train/train/tensor_param_model_layers_37_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/max_abs":0.000186920166015625,"train/train/tensor_param_model_layers_74_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/mean":-6.389617919921875e-05,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/max_abs":0.140625,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/norm":4.875,"train/train/layer_model_layers_89/grad/std":8.884090212987747e-05,"train/train/tensor_grad_model_layers_55_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/norm":0.0007749142261283971,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/norm":0.013720595878643499,"train/train/tensor_act_model_layers_30_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_4/grad/norm":0.09965629962840196,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/mean":-5.5122654885053635e-08,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/max_abs":0.0012664794921875,"train/train/tensor_grad_model_layers_69_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_mlp_waleed_W_g_weight/mean":2.9087066650390625e-05,"train/train/tensor_param_model_layers_72_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0/std":2.539074576731993,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_23/param/mean":0.0015320443139991225,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/std":0.0537109375,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/max_abs":0.25,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/mean":3.457069396972656e-05,"train/train/tensor_act_model_layers_19_mlp/norm":371.2307213999571,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/max_abs":0.002532958984375,"train/train/tensor_act_model_layers_22_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/norm":4.84375,"train/train/tensor_grad_model_layers_40_mlp_waleed_W_u_weight/mean":2.0116567611694336e-07,"train/train/tensor_act_model_layers_38_mlp_down_proj/std":0.05383349134977785,"train/train/tensor_param_model_layers_17_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_mlp_waleed_W_u_weight/std":0.03125,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/mean":-8.188188076019287e-06,"train/train/tensor_act_model_layers_69_self_attn_v_proj/std":0.46337972375618763,"train/train/layer_model_layers_1/act/norm":32438.988686712968,"train/train/tensor_act_model_layers_55/std":2.5937752712216797,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/norm":0.0136162443507379,"train/train/tensor_act_model_layers_14_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_mlp/max_abs":4.90625,"train/train/tensor_act_model_layers_4_mlp_waleed_W_u/std":0.3632813753930495,"train/train/tensor_act_model_layers_9/norm":17768.43962162453,"train/train/tensor_grad_model_layers_59_mlp_waleed_W_u_weight/max_abs":0.0007171630859375,"train/train/tensor_act_model_layers_18_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/std":0.0224609375,"train/train/tensor_act_model_layers_69_mlp_waleed_W_g/std":0.421875462280678,"train/train/tensor_act_model_layers_35_post_attention_layernorm/mean":0.0026760101318359375,"train/train/tensor_act_model_layers_88_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/mean":4.04752790927887e-06,"train/train/tensor_act_model_layers_55_mlp_down_proj/std":0.08557225491694422,"train/train/tensor_act_model_layers_18_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38/max_abs":25.125,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/mean":7.268041372299194e-06,"train/train/tensor_act_model_layers_15_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/norm":5.125,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/max_abs":0.0004215240478515625,"train/train/tensor_act_model_layers_54_self_attn_k_proj/mean":0.0482177734375,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/norm":5.03125,"train/train/layer_model_layers_75/act/norm":21640.195257388514,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/norm":0.013572364790297943,"train/train/tensor_grad_model_layers_39_mlp_waleed_W_g_weight/norm":0.013733791746876549,"train/train/tensor_grad_model_layers_22_mlp_waleed_W_g_weight/mean":-1.1862721294164658e-07,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/max_abs":0.2197265625,"train/train/tensor_act_model_layers_3_self_attn/norm":221.3439311234897,"train/train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer__model_layers_57/param/max_abs":1,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/std":0.034423828125,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/mean":8.463859558105469e-06,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp/mean":0.0003580152988433838,"train/train/tensor_grad_model_layers_28_mlp_waleed_W_g_weight/mean":6.070331437513232e-08,"train/train/layer__model_layers_61/param/std":0.05175450399818072,"train/train/layer_model_layers_82/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_waleed_W_g_weight/max_abs":0.12353515625,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_q_proj/mean":-0.0007772445678710938,"train/train/tensor_act_model_layers_5_mlp_waleed/mean":-0.0029754638671875,"train/train/tensor_act_model_layers_17_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_k_proj/mean":0.04022216796875,"train/train/tensor_act_model_layers_55_mlp_down_proj/max_abs":0.8828125,"train/train/layer__model_layers_27/param/norm":19.60039012287128,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/norm":0.0031125031130604707,"train/train/tensor_act_model_layers_6_self_attn_q_proj/mean":-0.042724609375,"train/train/layer_model_layers_43/act/max_abs":24.75,"train/train/tensor_act_model_layers_79_self_attn_o_proj/mean":0.0008535385131835938,"train/train/tensor_act_model_layers_59_mlp_waleed/mean":-0.00122833251953125,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_39_self_attn/max_abs":1.6171875,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/std":6.0981278842597325e-05,"train/train/tensor_act_model_layers_31_mlp_waleed/mean":0.0011043548583984375,"train/train/tensor_act_model_layers_83_self_attn_o_proj/std":0.3339844865408371,"train/train/tensor_act_model_layers_1_mlp_waleed_W_u/std":0.9209042561085313,"train/train/tensor_act_model_layers_70_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_k_proj/std":0.9638687433792589,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_70_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75/mean":0.0139923095703125,"train/train/tensor_act_model_layers_9_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/mean":1.0423362255096436e-05,"train/train/tensor_act_model_layers_4_mlp_waleed_W_u/max_abs":2.5625,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/norm":3.4375,"train/train/tensor_act_model_layers_75_mlp_waleed_W_u/mean":-0.001415252685546875,"train/train/tensor_act_model_layers_87_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_26/act/mean":-0.000415802001953125,"train/train/tensor_act_model_layers_60_mlp_waleed_W_u/norm":3061.565939494351,"train/train/tensor_param_model_layers_44_mlp_waleed_W_g_weight/max_abs":0.1328125,"train/train/tensor_act_model_layers_85_input_layernorm/norm":5792.607421878005,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/norm":8,"train/train/tensor_act_model_layers_68_self_attn_o_proj/max_abs":1.4765625,"train/train/tensor_param_model_layers_37_mlp_waleed_W_u_weight/norm":4.625,"train/train/tensor_act_model_layers_68_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp/mean":0.006103515625,"train/train/tensor_act_model_layers_53_self_attn_v_proj/mean":0.0007915496826171875,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/std":0.03857421875,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/norm":3.921875,"train/train/tensor_act_model_layers_2_mlp_down_proj/std":0.5947418028404788,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/mean":2.407468855381012e-07,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/std":0.0233154296875,"train/train/tensor_param_model_layers_21_mlp_waleed_W_u_weight/mean":-8.440017700195312e-05,"train/train/tensor_act_model_layers_4_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/max_abs":0.111328125,"train/train/tensor_act_model_layers_55_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_waleed_W_g_weight/norm":4.4375,"train/train/tensor_act_model_layers_65_input_layernorm/mean":0.0046844482421875,"train/train/tensor_act_model_layers_37_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_waleed_W_u_weight/max_abs":0.1689453125,"train/train/tensor_param_model_layers_89_mlp_waleed_W_g_weight/mean":0.00010204315185546875,"train/train/tensor_grad_model_norm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/mean":-3.2334355637431145e-08,"train/train/tensor_param_model_layers_42_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/std":0.00103759765625,"train/train/tensor_act_model_layers_80_mlp_waleed_W_g/norm":4617.777606407355,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/mean":0.0002727508544921875,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/max_abs":0.00021266937255859375,"train/train/tensor_act_model_layers_31_self_attn_v_proj/mean":0.00377655029296875,"train/train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/std":7.331719027789136e-05,"train/train/layer__model_layers_87/param/max_abs":1,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/max_abs":0.1328125,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/std":0.03857421875,"train/train/tensor_act_model_layers_60_input_layernorm/max_abs":4.90625,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/norm":6.71875,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/norm":0.023318305512268704,"train/train/tensor_act_model_layers_60_self_attn_k_proj/norm":4822.336515410158,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/max_abs":0.00043487548828125,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_input_layernorm/mean":0.0029726028442382812,"train/train/tensor_act_model_layers_83_self_attn_k_proj/max_abs":7.03125,"train/train/tensor_act_model_layers_55_mlp_waleed/mean":-0.0005178451538085938,"train/train/tensor_act_model_layers_43_self_attn_o_proj/max_abs":1.828125,"train/train/layer_model_layers_8/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/mean":0.0003871917724609375,"train/train/tensor_act_model_layers_5_self_attn/mean":0.001613616943359375,"train/train/layer__model_layers_42/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/norm":4.28125,"train/train/tensor_act_model_layers_26_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/std":0.0238037109375,"train/train/tensor_act_model_layers_56_mlp_waleed_W_g/std":0.35742208019625915,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/mean":1.2229429557919502e-07,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/norm":1422.8903211425861,"train/train/tensor_act_model_layers_58_self_attn_v_proj/mean":-0.0054779052734375,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn_v_proj/max_abs":5,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/norm":6.375,"train/train/tensor_act_model_layers_66_mlp_waleed/max_abs":3.859375,"train/train/layer_model_layers_12/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58/mean":0.00791168212890625,"train/train/tensor_act_model_layers_53_self_attn/max_abs":1.9375,"train/train/layer_model_layers_22/grad/std":5.438716485349396e-05,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/std":3.745886363901723e-05,"train/train/tensor_act_model_layers_53_post_attention_layernorm/std":1.0000012073892992,"train/train/tensor_grad_model_layers_26_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_k_proj/norm":5165.9600566307,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/mean":-1.780688762664795e-06,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/std":0.038818359375,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/max_abs":0.0859375,"train/train/tensor_param_model_layers_18_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_input_layernorm/mean":-0.00563812255859375,"train/train/tensor_act_model_layers_85_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/max_abs":0.000820159912109375,"train/train/tensor_act_model_layers_50_self_attn_o_proj/max_abs":1.375,"train/train/layer__model_layers_10/param/std":0.0478137773193981,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/norm":0.012013038314662177,"train/train/tensor_param_model_layers_75_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_39_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_46_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_o_proj/norm":1349.676532873374,"train/train/layer__model_layers_64/param/max_abs":1,"train/train/tensor_act_model_layers_12_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/std":3.538775533307867e-05,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81/std":3.1992262512976115,"train/train/tensor_act_model_layers_55_input_layernorm/max_abs":5.0625,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/mean":0.00024318695068359375,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/max_abs":4.65625,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_84/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn/mean":-0.00021195411682128906,"train/train/tensor_param_model_layers_11_mlp_waleed_W_u_weight/max_abs":0.12060546875,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/max_abs":0.00194549560546875,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/std":8.383025357480188e-05,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/max_abs":0.0003376007080078125,"train/train/tensor_act_model_layers_57_post_attention_layernorm/max_abs":4.96875,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/max_abs":0.162109375,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/max_abs":0.001495361328125,"train/train/tensor_param_model_layers_88_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_11/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_61_self_attn_v_proj/norm":2248.343358790184,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/std":9.256340736702327e-05,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_71/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/max_abs":0.000247955322265625,"train/train/tensor_act_model_layers_18_mlp_waleed/norm":786.8752070195026,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/norm":0.0010318474051803862,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_79/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer_model_layers_71/grad/norm":0.044707968942052416,"train/train/layer_model_layers_63/act/mean":0.010042130947113037,"train/train/tensor_act_model_layers_4_mlp_waleed/norm":1904.841824483878,"train/train/tensor_act_model_layers_74_self_attn/std":0.3930698799637309,"train/train/layer__model_layers_76/param/max_abs":1,"train/train/tensor_act_model_layers_56_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_input_layernorm/max_abs":5.03125,"train/train/tensor_grad_model_layers_69_mlp_waleed_W_g_weight/norm":0.019748528214433507,"train/train/tensor_act_model_layers_68_mlp/mean":-0.0004944801330566406,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/std":2.8803708921677548e-05,"train/train/tensor_act_model_layers_68_self_attn_k_proj/mean":-0.02227783203125,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp_down_proj/norm":13242.716038913999,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/std":3.437118904562111e-05,"train/train/layer_model_layers_58/act/mean":0.00725129060447216,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/max_abs":0.000270843505859375,"train/train/tensor_act_model_layers_70_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/norm":0.003081312123469371,"train/train/tensor_act_model_layers_77_self_attn_o_proj/norm":2034.2776746576142,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/norm":4.6875,"train/train/tensor_act_model_layers_84_mlp_waleed_W_g/max_abs":3.859375,"train/train/layer__model_layers_70/param/mean":0.001580201147499025,"train/train/tensor_act_model_layers_72_self_attn_o_proj/std":0.28662261158282965,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/std":4.2758702239580606e-05,"train/train/layer_model_layers_30/grad/mean":6.43980138689605e-08,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_k_proj/std":0.9677759079319421,"train/train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_41/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_o_proj/max_abs":1.640625,"train/train/tensor_act_model_layers_11_self_attn_k_proj/std":1.1738333332046194,"train/train/tensor_act_model_layers_22_mlp/norm":429.6958412000396,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/norm":0.013935610711553649,"train/train/layer__model_layers_39/param/mean":0.0015670847781176873,"train/train/tensor_act_model_layers_20_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/mean":-6.731599569320679e-06,"train/train/tensor_act_model_layers_25_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_mlp_waleed_W_u_weight/norm":0.019989829010387185,"train/train/tensor_param_model_layers_63_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_66_self_attn_k_proj/norm":5116.489561100741,"train/train/tensor_param_model_layers_82_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/std":1.9005024453476426e-05,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_k_proj/norm":6240.823881535988,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/std":0.047607421875,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/std":0.044677734375,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/mean":-3.790855407714844e-05,"train/train/tensor_act_model_layers_13_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_mlp_waleed_W_g_weight/mean":1.428648829460144e-06,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/norm":0.0020511758799287706,"train/train/tensor_act_model_layers_6_input_layernorm/norm":5792.609375007697,"train/train/tensor_act_model_layers_55_self_attn_q_proj/mean":0.025848388671875,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_k_proj/max_abs":7,"train/train/tensor_act_model_layers_38_self_attn/std":0.23706407586112035,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/norm":0.004523040717845784,"train/train/tensor_act_model_layers_88_input_layernorm/std":1.0000001099178855,"train/train/tensor_grad_model_layers_8_mlp_waleed_W_g_weight/max_abs":0.0022125244140625,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/norm":0.008913327299619822,"train/train/tensor_act_model_layers_64_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_68/param/max_abs":1,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/mean":-3.910064697265625e-05,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/mean":-2.0079314708709717e-06,"train/train/tensor_act_model_layers_52_mlp_waleed_W_u/max_abs":2.390625,"train/train/tensor_act_model_layers_23_input_layernorm/max_abs":5.1875,"train/train/tensor_act_model_layers_87_self_attn/norm":3496.356921138058,"train/train/layer_model_layers_15/grad/std":5.271406250803264e-05,"train/train/tensor_act_model_layers_16_mlp_waleed_W_u/std":0.24389686336239685,"train/train/layer_model_layers_24/grad/std":6.006050934056645e-05,"train/train/layer_model_layers_9/act/norm":22498.43756242645,"train/train/tensor_act_model_layers_58_mlp_down_proj/std":0.08215360426990974,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn/mean":-0.004009246826171875,"train/train/tensor_param_model_layers_30_mlp_waleed_W_g_weight/mean":-9.679794311523438e-05,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/norm":5.3125,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_mlp_waleed_W_g_weight/std":5.460802494143194e-05,"train/train/tensor_param_model_layers_28_mlp_waleed_W_g_weight/norm":4.40625,"train/train/tensor_act_model_layers_15_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_mlp_waleed_W_u/norm":2945.366306073896,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/norm":0.017879238253097274,"train/train/tensor_param_model_layers_90_mlp_waleed_W_g_weight/std":0.05322265625,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/max_abs":0.21484375,"train/train/layer__model_layers_61/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp_waleed/mean":-0.0009317398071289062,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/norm":4.9375,"train/train/tensor_grad_model_layers_90_mlp_waleed_W_u_weight/mean":-4.997709766030312e-07,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_mlp_waleed_W_u_weight/mean":0.00037384033203125,"train/train/tensor_act_model_layers_83_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16/mean":-0.00666046142578125,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/std":3.409838022361572e-05,"train/train/tensor_param_model_layers_20_mlp_waleed_W_g_weight/norm":4.21875,"train/train/tensor_act_model_layers_52_mlp_waleed/mean":0.0003027915954589844,"train/train/tensor_act_model_layers_72_mlp_waleed_W_g/mean":-0.00376129150390625,"train/train/tensor_act_model_layers_31_self_attn/mean":0.00020304322242736816,"train/train/tensor_param_model_layers_41_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn_v_proj/norm":3518.2329893172923,"train/train/tensor_grad_model_layers_77_mlp_waleed_W_u_weight/mean":3.641471266746521e-07,"train/train/tensor_act_model_layers_58_post_attention_layernorm/mean":0.0019265413284301758,"train/train/tensor_act_model_layers_65_mlp_down_proj/max_abs":0.984375,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/mean":1.7552520148456097e-07,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/norm":6.5,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/norm":6.3125,"train/train/layer__model_layers_21/param/mean":0.001585866657322543,"train/train/tensor_param_model_layers_84_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_v_proj/max_abs":2.890625,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp_waleed/mean":-0.016876220703125,"train/train/tensor_act_model_layers_69_post_attention_layernorm/norm":5792.605346681801,"train/train/tensor_grad_model_layers_64_mlp_waleed_W_u_weight/norm":0.017582998659766605,"train/train/layer__model_layers_30/param/mean":0.0014369446848186427,"train/train/tensor_act_model_layers_44_mlp_waleed_W_g/std":0.3242187574505805,"train/train/tensor_param_model_layers_73_mlp_waleed_W_g_weight/norm":6.40625,"train/train/tensor_param_model_layers_2_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/mean":-1.913309097290039e-05,"train/train/tensor_param_model_layers_38_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_70_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_4_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_o_proj/max_abs":2.5625,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/mean":8.833594620227814e-07,"train/train/tensor_act_model_layers_32_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/std":3.99403475370155e-05,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/mean":0.000316619873046875,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/std":1.7500785056361548e-05,"train/train/tensor_act_model_layers_56_self_attn_q_proj/mean":-0.0982666015625,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/mean":-0.0004329681396484375,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_83_mlp_waleed_W_u_weight/std":0.041259765625,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/norm":0.0337773208944544,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/max_abs":0.11669921875,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/std":2.6271799058410578e-05,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/mean":4.8748916015028954e-08,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/mean":9.202957153320312e-05,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/mean":-5.713663995265961e-07,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/max_abs":0.000804901123046875,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/norm":3.3125,"train/train/tensor_act_model_layers_44_post_attention_layernorm/norm":5792.614135747521,"train/train/tensor_act_model_layers_6_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_waleed_W_u/max_abs":2.234375,"train/train/tensor_act_model_layers_34_self_attn_o_proj/norm":1015.0820434750545,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/norm":0.01702979300330042,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/max_abs":0.0004329681396484375,"train/train/tensor_param_model_layers_35_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp_waleed_W_u/mean":0.03765869140625,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_waleed_W_u/norm":3725.8964435537064,"train/train/tensor_act_model_layers_46_input_layernorm/max_abs":5.125,"train/train/tensor_grad_model_layers_11_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp/mean":-0.0014400482177734375,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_mlp_waleed_W_u/max_abs":2.09375,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/mean":-1.6689300537109375e-06,"train/train/tensor_act_model_layers_29_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/norm":0.03429046119032704,"train/train/layer_model_layers_46/grad/std":4.165308054600705e-05,"train/train/layer__model_layers_73/param/max_abs":1,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/max_abs":0.00019168853759765625,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/norm":0.0009126299469175911,"train/train/tensor_act_model_layers_35_self_attn_k_proj/std":0.9697282493257131,"train/train/layer__model_layers_30/param/std":0.048768937638729054,"train/train/tensor_act_model_layers_46_mlp_waleed_W_g/mean":-0.0086517333984375,"train/train/tensor_act_model_layers_42_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/layer_model_layers_78/grad/max_abs":0.0023193359375,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/norm":0.026891302996815372,"train/train/tensor_act_model_layers_71/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_mlp_waleed_W_u_weight/std":3.910541080980197e-05,"train/train/tensor_grad_model_layers_19_mlp_waleed_W_u_weight/max_abs":0.000804901123046875,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/std":1.2495603470157361e-05,"train/train/tensor_act_model_layers_46_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp_waleed_W_u/norm":2300.9246379564624,"train/train/tensor_param_model_layers_87_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_waleed_W_g_weight/mean":1.7182901501655579e-07,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/max_abs":0.1552734375,"train/train/layer_model_layers_12/act/std":0.9355264578425265,"train/train/tensor_act_model_layers_73_post_attention_layernorm/norm":5792.612426758259,"train/train/tensor_act_model_layers_78_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_mlp_waleed_W_u/max_abs":2.5625,"train/train/tensor_act_model_layers_91_mlp_waleed/max_abs":13.25,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/max_abs":0.224609375,"train/train/layer__model_layers_64/param/norm":21.280355157398102,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/std":1.6683066551792132e-05,"train/train/layer__model_layers_18/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/mean":-1.424551010131836e-05,"train/train/layer_model_layers_10/act/norm":23290.402039051158,"train/train/tensor_act_model_layers_30_self_attn_o_proj/max_abs":3.203125,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/max_abs":0.23046875,"train/train/tensor_act_model_layers_3_mlp_waleed/norm":3486.969157920422,"train/train/tensor_act_model_layers_62_self_attn_v_proj/mean":0.0085601806640625,"train/train/tensor_act_model_layers_12_mlp_waleed_W_u/norm":2072.527727928834,"train/train/tensor_param_model_layers_68_mlp_waleed_W_u_weight/max_abs":0.1845703125,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/max_abs":0.11669921875,"train/train/tensor_act_model_layers_20_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/mean":-2.5510787963867188e-05,"train/train/tensor_grad_model_layers_66_mlp_waleed_W_g_weight/max_abs":0.000827789306640625,"train/train/tensor_param_model_layers_6_mlp_waleed_W_g_weight/std":0.0234375,"train/train/tensor_act_model_layers_65_self_attn_k_proj/mean":-0.028228759765625,"train/train/tensor_act_model_layers_27_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_17/grad/norm":0.05711666255149961,"train/train/tensor_act_model_layers_18_self_attn/max_abs":1.2265625,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/std":3.827182640242921e-05,"train/train/tensor_act_model_layers_60_mlp_waleed_W_u/std":0.3735362641935186,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/max_abs":0.00067138671875,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/max_abs":0.193359375,"train/train/tensor_grad_model_layers_21_mlp_waleed_W_g_weight/norm":0.01679321071666058,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/norm":0.0005049910374592966,"train/train/tensor_act_model_layers_68_mlp_waleed_W_u/mean":-0.0096893310546875,"train/train/tensor_grad_model_layers_90_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_mlp_waleed_W_g_weight/max_abs":0.1220703125,"train/train/tensor_act_model_layers_4_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_input_layernorm/max_abs":5.4375,"train/train/tensor_act_model_layers_80_self_attn_k_proj/mean":-0.040771484375,"train/train/layer__model_layers_29/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp_waleed/max_abs":7.5625,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/std":0.0245361328125,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/std":0.0380859375,"train/train/tensor_grad_model_layers_36_mlp_waleed_W_u_weight/max_abs":0.0009307861328125,"train/train/layer_model_layers_37/grad/norm":0.03421704584826836,"train/train/tensor_act_model_layers_23_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_mlp_waleed_W_u_weight/max_abs":0.0010528564453125,"train/train/tensor_act_model_layers_89_self_attn/norm":2306.719301150095,"train/train/tensor_grad_model_layers_41_mlp_waleed_W_u_weight/norm":0.014511581718034012,"train/train/tensor_param_model_layers_73_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/norm":0.017334199286189744,"train/train/tensor_act_model_layers_91_input_layernorm/mean":-0.00421142578125,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/norm":0.0006762055296169577,"train/train/tensor_param_model_layers_78_mlp_waleed_W_g_weight/mean":0.0002346038818359375,"train/train/tensor_param_model_layers_19_mlp_waleed_W_u_weight/norm":4.21875,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/mean":-2.171844244003296e-06,"train/train/layer_model_layers_55/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/std":3.169858722738326e-05,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/mean":-0.000278472900390625,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_down_proj/norm":1755.3542312210461,"train/train/tensor_act_model_layers_10_self_attn_k_proj/norm":9367.722183424652,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/std":4.718718744069227e-05,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/norm":0.003091966859310032,"train/train/tensor_act_model_layers_56/norm":14992.82160293812,"train/train/tensor_act_model_layers_59_mlp_down_proj/std":0.08215359973647395,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/max_abs":0.0002994537353515625,"train/train/tensor_param_model_layers_62_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_39_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/max_abs":0.2333984375,"train/train/tensor_param_model_layers_61_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/max_abs":0.1650390625,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/norm":4.40625,"train/train/tensor_act_model_layers_6_self_attn/mean":0.00013178586959838867,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/mean":-1.6042031347751617e-07,"train/train/tensor_act_model_layers_78/mean":0.019622802734375,"train/train/tensor_grad_model_layers_73_mlp_waleed_W_g_weight/std":6.486137768229566e-05,"train/train/tensor_act_model_layers_18_mlp/mean":0.0003600120544433594,"train/train/tensor_act_model_layers_39_self_attn_v_proj/norm":2559.5838915323125,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/max_abs":0.10595703125,"train/train/tensor_act_model_layers_38/std":2.6406496504578785,"train/train/tensor_act_model_layers_26_self_attn/max_abs":1.203125,"train/train/tensor_act_model_layers_11_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33/max_abs":26.375,"train/train/tensor_act_model_layers_79_mlp_waleed_W_g/norm":4228.476330778228,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/max_abs":0.1474609375,"train/train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn/max_abs":0.515625,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/max_abs":0.2294921875,"train/train/tensor_act_model_layers_1_post_attention_layernorm/std":1.0000000245636327,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/norm":6.875,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_89/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/max_abs":0.000270843505859375,"train/train/tensor_act_model_layers_89_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_input_layernorm/frac_near_dtype_limit":0,"train/epoch":0.8088433540037746,"train/train/layer_model_layers_15/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_waleed/norm":783.4970282134266,"train/train/tensor_act_model_layers_29_self_attn_k_proj/std":0.9443375879369946,"train/train/tensor_act_model_layers_44_input_layernorm/max_abs":5.1875,"train/train/tensor_act_model_layers_3_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_input_layernorm/norm":5792.607177741933,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/max_abs":0.000518798828125,"train/train/tensor_act_model_layers_63_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/mean":-4.149042069911957e-07,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/std":4.135981504888169e-05,"train/train/layer__model_layers_40/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_mlp_waleed_W_u_weight/mean":0.00113677978515625,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/mean":-3.0977389542385936e-09,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_52/param/norm":20.00065916882491,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/mean":1.4045508578419685e-07,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_self_attn_o_proj/norm":635.5576996376487,"train/train/tensor_act_model_layers_73_mlp_waleed_W_u/mean":0.009857177734375,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/max_abs":0.2734375,"train/train/tensor_act_model_layers_4_mlp_waleed_W_u/norm":2970.7270146094706,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_42_mlp_waleed_W_u/std":0.3046875107699096,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_6/norm":18717.345059206604,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_q_proj/norm":5242.738492274336,"train/train/layer__model_layers_12/param/mean":0.0016036487406761895,"train/train/layer_model_layers_19/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/norm":0.0007258808597465022,"train/train/tensor_grad_model_layers_64_mlp_waleed_W_g_weight/mean":-2.8009526431560516e-07,"train/train/layer__model_layers_70/param/norm":21.959285167739978,"train/train/tensor_param_model_layers_63_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_89/mean":-0.0186767578125,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/std":1.728216625863208e-05,"train/train/tensor_act_model_layers_83_self_attn_k_proj/norm":6753.079501905838,"train/train/tensor_act_model_layers_48_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_mlp_waleed_W_g_weight/std":0.037841796875,"train/train/layer_model_layers_53/grad/std":4.304334606603799e-05,"train/train/tensor_param_model_layers_77_mlp_waleed_W_u_weight/max_abs":0.1787109375,"train/train/layer_model_layers_82/grad/max_abs":0.00139617919921875,"train/train/tensor_act_model_layers_62_mlp_waleed/norm":1193.0477630650116,"train/train/tensor_act_model_layers_31_post_attention_layernorm/norm":5792.60961914522,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/std":4.442095248573097e-05,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/mean":-9.167194366455078e-05,"train/train/tensor_act_model_layers_35_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp/max_abs":0.66015625,"train/train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/norm":0.0008966689156205377,"train/train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp_waleed_W_u/max_abs":2.765625,"train/train/tensor_act_model_layers_44_mlp_down_proj/norm":423.02089659152307,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/std":0.037109375,"train/train/tensor_grad_model_layers_70_mlp_waleed_W_g_weight/norm":0.019897648161024843,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_v_proj/mean":0.00624847412109375,"train/train/layer__model_layers_35/param/max_abs":1,"train/train/tensor_act_model_layers_17_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_o_proj/norm":414.039810164717,"train/train/tensor_param_model_layers_18_mlp_waleed_W_u_weight/norm":4.21875,"train/train/tensor_act_model_layers_63/norm":15053.635626405578,"train/train/tensor_act_model_layers_67_self_attn/std":0.3168956951040336,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/std":0.00032808353368179473,"train/train/tensor_param_model_layers_87_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/mean":1.4221313904272392e-08,"train/train/tensor_grad_model_layers_63_mlp_waleed_W_g_weight/mean":-1.069856807589531e-07,"train/train/tensor_act_model_layers_49_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_waleed_W_g_weight/max_abs":0.0005035400390625,"train/train/tensor_act_model_layers_33_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp/mean":-0.00017070770263671875,"train/train/tensor_act_model_layers_78_self_attn/std":0.2670985080762469,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/mean":0.0001735687255859375,"train/train/layer_model_layers_1/grad/std":0.0006381036443654138,"train/train/layer__model_layers_10/param/mean":0.0016562332415171607,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/mean":-4.744529724121094e-05,"train/train/layer__model_layers_20/param/std":0.04794314736019825,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp_waleed/mean":-0.001407623291015625,"train/train/tensor_grad_model_layers_24_mlp_waleed_W_u_weight/mean":-5.4977135732769966e-08,"train/train/tensor_act_model_layers_90_self_attn_k_proj/mean":-0.0985107421875,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/max_abs":0.0005645751953125,"train/train/tensor_act_model_layers_18_input_layernorm/mean":-0.00690460205078125,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/std":0.038330078125,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/max_abs":0.328125,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_mlp_waleed_W_g_weight/std":5.0674196723906265e-05,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/norm":0.0007114129150394902,"train/train/tensor_grad_model_layers_29_mlp_waleed_W_g_weight/max_abs":0.000579833984375,"train/train/tensor_grad_model_layers_43_mlp_waleed_W_u_weight/mean":-2.0326115190982819e-07,"train/train/tensor_act_model_layers_65_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/layer_model_layers_41/act/norm":18971.242836327005,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/max_abs":0.271484375,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/mean":0.0001983642578125,"train/train/tensor_grad_model_layers_47_mlp_waleed_W_g_weight/std":3.576642367491493e-05,"train/train/tensor_act_model_layers_10_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/max_abs":0.1396484375,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/std":0.03369140625,"train/train/tensor_act_model_layers_59_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/norm":7.1875,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/norm":0.005577332664764071,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/norm":0.01985979317765814,"train/train/tensor_act_model_layers_32_mlp_down_proj/std":0.0721438687962344,"train/train/tensor_act_model_layers_76_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/norm":0.01864631515462905,"train/train/tensor_act_model_layers_69_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_92/grad/norm":0.09574476255917293,"train/train/tensor_act_model_layers_24_input_layernorm/norm":5792.607666018264,"train/train/tensor_act_model_layers_30_mlp_waleed/frac_near_user_limit":0,"train/train/layer__model_layers_81/param/mean":0.0016323571644037637,"train/train/tensor_grad_model_layers_20_mlp_waleed_W_g_weight/std":3.9406438820417634e-05,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/mean":-5.841255187988281e-05,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/mean":2.8777867555618286e-07,"train/train/tensor_act_model_layers_6_self_attn_k_proj/std":1.7343759836609463,"train/train/layer__model_layers_4/param/mean":0.0015928057166231962,"train/train/tensor_act_model_layers_93_mlp_waleed/mean":-0.02410888671875,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/max_abs":0.2080078125,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_mlp_waleed/max_abs":6.03125,"train/train/tensor_act_model_layers_28_input_layernorm/max_abs":5.59375,"train/train/tensor_act_model_layers_5/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/std":0.0291748046875,"train/train/tensor_act_model_layers_63_mlp_down_proj/max_abs":0.8203125,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp_waleed_W_u/max_abs":6.40625,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/max_abs":0.1474609375,"train/train/tensor_act_model_layers_10_mlp_waleed/std":0.15844786222136906,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/std":0.04541015625,"train/train/tensor_act_model_layers_48_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/mean":-2.577435225248337e-07,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/max_abs":0.00011157989501953125,"train/train/tensor_act_model_layers_67_self_attn_q_proj/norm":7210.638806646845,"train/train/tensor_param_model_layers_53_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_40/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_82/grad/std":9.510109279040106e-05,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_v_proj/std":0.7246120192484022,"train/train/tensor_act_model_layers_28_mlp_waleed_W_g/max_abs":3.140625,"train/train/tensor_act_model_layers_31_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn/norm":414.039810164717,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_41/param/mean":0.0017125550744686037,"train/train/layer__model_layers_82/param/norm":23.803558032361465,"train/train/tensor_act_model_layers_14_self_attn/norm":437.36988984304406,"train/train/tensor_act_model_layers_84_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_mlp_down_proj/norm":480.87008708016845,"train/train/tensor_grad_model_layers_30_mlp_waleed_W_u_weight/mean":-1.1021620593965054e-07,"train/train/tensor_act_model_layers_84_mlp_waleed/std":0.4257834975403121,"train/train/tensor_act_model_layers_24_self_attn_v_proj/max_abs":3.25,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/std":3.760589240370979e-05,"train/train/tensor_act_model_layers_34_mlp_waleed_W_u/mean":-0.0010251998901367188,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/std":0.0223388671875,"train/train/tensor_act_model_layers_88_mlp_down_proj/mean":0.00653076171875,"train/train/tensor_act_model_layers_41_post_attention_layernorm/mean":0.002880096435546875,"train/train/tensor_param_model_layers_10_mlp_waleed_W_u_weight/norm":4.1875,"train/train/tensor_act_model_layers_78_mlp_waleed/max_abs":5.4375,"train/train/tensor_act_model_layers_68_mlp_down_proj/mean":-0.0004944801330566406,"train/train/tensor_act_model_layers_35/max_abs":25.875,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_mlp_waleed_W_g_weight/norm":4.625,"train/train/tensor_act_model_layers_82_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_self_attn/std":0.10974322754416757,"train/train/tensor_param_model_layers_30_mlp_waleed_W_u_weight/max_abs":0.1337890625,"train/train/tensor_param_model_layers_57_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/layer_model_layers_3/act/std":1.0279831761724016,"train/train/tensor_param_model_layers_32_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/std":0.026123046875,"train/train/tensor_act_model_layers_41_self_attn_k_proj/mean":0.0086822509765625,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_mlp_waleed_W_u_weight/max_abs":0.1689453125,"train/train/tensor_grad_model_layers_39_mlp_waleed_W_g_weight/std":3.791131042381527e-05,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/norm":3.03125,"train/train/tensor_grad_model_layers_86_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_mlp_waleed_W_g_weight/std":0.056884765625,"train/train/tensor_act_model_layers_76_self_attn/mean":-0.00035762786865234375,"train/train/tensor_act_model_layers_26_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn/norm":1682.9537174894388,"train/train/tensor_act_model_layers_68_input_layernorm/mean":0.00565338134765625,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_mlp_waleed_W_g_weight/norm":0.01393196797811909,"train/train/tensor_grad_model_layers_6_mlp_waleed_W_g_weight/norm":0.04102180071158768,"train/train/layer_model_layers_13/act/norm":21859.355017545968,"train/train/layer__model_layers_86/param/norm":24.841352085534716,"train/train/tensor_param_model_layers_55_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/std":0.04931640625,"train/train/tensor_act_model_layers_82_self_attn_k_proj/norm":6695.031208648449,"train/train/tensor_act_model_layers_30_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_waleed_W_u_weight/max_abs":0.0005645751953125,"train/train/tensor_act_model_layers_35_input_layernorm/std":1.000000446241657,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/mean":-0.00022125244140625,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/norm":0.018751505950214364,"train/train/tensor_act_model_layers_44_mlp_waleed_W_g/mean":-0.00708770751953125,"train/train/tensor_param_model_layers_54_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn_o_proj/norm":437.36988984304406,"train/train/tensor_grad_model_layers_32_mlp_waleed_W_g_weight/norm":0.015641352917617576,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/mean":4.224013537168503e-06,"train/train/layer_model_layers_3/grad/max_abs":0.0057373046875,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/norm":5.5,"train/train/tensor_act_model_layers_39_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_54_mlp_down_proj/mean":0.0002770423889160156,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/max_abs":0.014892578125,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_mlp_waleed_W_g_weight/max_abs":0.00078582763671875,"train/train/tensor_act_model_layers_56_self_attn_q_proj/max_abs":5.125,"train/train/tensor_act_model_layers_24_post_attention_layernorm/std":1.0000000818545198,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/norm":0.02108629942813394,"train/train/tensor_act_model_layers_43/max_abs":24.75,"train/train/layer__model_layers_29/param/max_abs":1,"train/train/tensor_act_model_layers_22/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_o_proj/max_abs":3.71875,"train/train/tensor_act_model_layers_47_self_attn_q_proj/max_abs":3.859375,"train/train/tensor_param_model_layers_91_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_18_mlp_waleed_W_u_weight/mean":2.1969981389702298e-08,"train/train/tensor_act_model_layers_16_mlp/norm":336.7701577860633,"train/train/tensor_param_model_layers_19_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_o_proj/norm":529.0076487260571,"train/train/tensor_param_model_layers_90_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_64/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_mlp_waleed_W_u_weight/std":4.2919706725551446e-05,"train/train/tensor_act_model_layers_88_post_attention_layernorm/max_abs":6.25,"train/train/tensor_act_model_layers_21_self_attn_q_proj/mean":0.0124359130859375,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/norm":6.71875,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_k_proj/mean":0.0154571533203125,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/std":0.0230712890625,"train/train/layer__model_layers_83/param/mean":0.0016935589532956318,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/std":0.046142578125,"train/train/tensor_param_model_layers_79_mlp_waleed_W_u_weight/std":0.038330078125,"train/train/tensor_act_model_layers_51_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_waleed_W_g_weight/norm":0.01584129825409805,"train/train/tensor_param_model_layers_42_mlp_waleed_W_u_weight/max_abs":0.1376953125,"train/train/tensor_act_model_layers_44_self_attn/mean":-0.00019657611846923828,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/std":0.03564453125,"train/train/tensor_act_model_layers_16_self_attn_v_proj/std":0.36523438318767004,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_91/grad/mean":1.3783138227909105e-07,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/std":0.038818359375,"train/train/tensor_act_model_layers_42_self_attn_k_proj/max_abs":4.6875,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_waleed/max_abs":3.90625,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/mean":-2.6240013539791107e-07,"train/train/layer_model_layers_81/act/norm":24510.31093433029,"train/train/tensor_act_model_layers_32_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/std":0.0272216796875,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/norm":7.59375,"train/train/tensor_grad_model_layers_74_mlp_waleed_W_u_weight/mean":-8.087590686045587e-08,"train/train/tensor_act_model_layers_53_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/norm":0.0022236904016206863,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/max_abs":0.00013065338134765625,"train/train/tensor_act_model_layers_74_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/std":0.044921875,"train/train/tensor_act_model_layers_45_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_post_attention_layernorm/mean":-0.0054931640625,"train/train/tensor_param_model_layers_60_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_13_mlp_waleed_W_u/norm":2155.54664115862,"train/train/tensor_grad_model_layers_28_mlp_waleed_W_u_weight/norm":0.014238240905202213,"train/train/tensor_act_model_layers_10_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_post_attention_layernorm/mean":-0.00676727294921875,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/mean":-1.27092789625749e-08,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/norm":0.010799816909425766,"train/train/tensor_act_model_layers_57_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_mlp_waleed_W_g_weight/norm":0.018464017484386818,"train/train/tensor_act_model_layers_69_mlp_down_proj/norm":724.7051076307085,"train/train/tensor_grad_model_layers_22_mlp_waleed_W_u_weight/norm":0.021641503464258836,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/max_abs":0.1416015625,"train/train/tensor_act_model_layers_21_mlp_waleed/max_abs":3.84375,"train/train/tensor_act_model_layers_12_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_19_input_layernorm/max_abs":5.21875,"train/train/tensor_act_model_layers_18_self_attn_q_proj/std":1.0312503341471968,"train/train/tensor_act_model_layers_23/max_abs":27,"train/train/tensor_param_model_embed_tokens_weight/norm":56.5,"train/train/tensor_act_model_layers_18_self_attn_o_proj/max_abs":1.2265625,"train/train/layer_model_layers_64/grad/std":6.106128181432023e-05,"train/train/tensor_param_model_layers_84_mlp_waleed_W_u_weight/max_abs":0.2412109375,"train/train/tensor_act_model_layers_18/frac_near_dtype_limit":0,"train/train/layer_model_layers_40/act/norm":18924.34304096097,"train/train/tensor_act_model_layers_23_mlp_waleed_W_g/norm":1916.6712449712775,"train/train/layer__model_layers_60/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp/std":0.07299846879094012,"train/train/tensor_act_model_layers_47_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/norm":4.3125,"train/train/tensor_act_model_layers_74_self_attn_k_proj/norm":6720.078730875044,"train/train/tensor_act_model_layers_90_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_self_attn_q_proj/std":0.9677791883988714,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/norm":0.0026922143185387306,"train/train/tensor_act_model_layers_84_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_input_layernorm/norm":5792.611328126523,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_16_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/std":0.0238037109375,"train/train/tensor_act_model_layers_29_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/std":0.00012875857243059346,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_24_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn_v_proj/max_abs":2.734375,"train/train/layer_model_layers_91/grad/std":0.00010608576484029719,"train/train/layer_model_layers_47/act/mean":-0.0010725905885919929,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/mean":1.7025740817189217e-09,"train/train/tensor_grad_model_layers_93_mlp_waleed_W_u_weight/mean":-2.9383227229118347e-07,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/norm":0.021486629021053652,"train/train/tensor_act_model_layers_50_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_mlp_waleed_W_u_weight/max_abs":0.000713348388671875,"train/train/tensor_act_model_layers_6_self_attn_q_proj/std":1.7089886910520276,"train/train/tensor_param_model_layers_74_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_52/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_waleed_W_u/std":0.36377053828708095,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn_v_proj/mean":0.00531768798828125,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/std":0.042236328125,"train/train/tensor_act_model_layers_78_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_k_proj/norm":7219.498409669028,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/std":5.444396549232393e-05,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/std":0.02978515625,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_86/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer_model_layers_31/act/mean":-0.0013221018016338348,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_waleed_W_g/max_abs":3.4375,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/std":0.059814453125,"train/train/tensor_act_model_layers_50_mlp_waleed_W_u/norm":2851.38098789432,"train/train/tensor_act_model_layers_35_self_attn_q_proj/norm":6253.262003499013,"train/train/tensor_act_model_layers_31_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_mlp/max_abs":1.0390625,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/max_abs":0.000514984130859375,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/norm":0.0013316899988868966,"train/train/layer_model_layers_12/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/mean":0.000118255615234375,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/std":0.0262451171875,"train/train/tensor_act_model_layers_67_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_waleed_W_u/std":0.46679702189175837,"train/train/tensor_act_model_layers_87_self_attn_q_proj/mean":0.170166015625,"train/train/tensor_param_model_layers_5_mlp_waleed_W_u_weight/mean":-0.00013637542724609375,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/norm":4.8125,"train/train/tensor_act_model_layers_93_input_layernorm/max_abs":5.9375,"train/train/tensor_act_model_layers_4_self_attn_k_proj/norm":8593.435648515806,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/max_abs":0.00021839141845703125,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/norm":7.5,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/std":4.869207181675383e-05,"train/train/tensor_act_model_layers_7_self_attn_q_proj/max_abs":5.75,"train/train/tensor_act_model_layers_70_self_attn_v_proj/norm":2687.9600861585736,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/mean":0.00011396408081054688,"train/train/layer_model_layers_47/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/max_abs":0.2734375,"train/train/tensor_grad_model_layers_75_mlp_waleed_W_u_weight/std":6.521313786496418e-05,"train/train/tensor_act_model_layers_4_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_post_attention_layernorm/max_abs":5.4375,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_66_mlp_waleed_W_g_weight/std":5.192262280310404e-05,"train/train/tensor_act_model_layers_61_mlp_waleed/std":0.15039064527138352,"train/train/tensor_act_model_layers_44_self_attn_q_proj/std":0.9052751174811691,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/norm":4.96875,"train/train/tensor_grad_model_layers_12_mlp_waleed_W_g_weight/mean":5.1106326282024384e-08,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/norm":0.004016990333957702,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/norm":0.002661554495622621,"train/train/tensor_grad_model_layers_57_mlp_waleed_W_u_weight/mean":-2.1629966795444489e-07,"train/train/tensor_param_model_layers_49_mlp_waleed_W_u_weight/std":0.0269775390625,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/norm":5,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_waleed_W_u/std":0.3403331110373092,"train/train/tensor_param_model_layers_18_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/mean":0.00018596649169921875,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_mlp_waleed_W_u_weight/std":8.764307432346274e-05,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp_down_proj/std":0.3066407095664509,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/norm":0.0457954061969594,"train/train/layer__model_layers_86/param/frac_near_user_limit":0,"train/train/layer__model_layers_28/param/max_abs":1,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_k_proj/mean":0.04925537109375,"train/train/tensor_act_model_layers_11_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/global/act/mean":-0.0391098111607182,"train/train/layer_model_layers_80/grad/norm":0.07468818718290328,"train/train/tensor_act_model_layers_54_self_attn/norm":585.6820045138335,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_v_proj/mean":-0.01715087890625,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/mean":-0.00015163421630859375,"train/train/tensor_act_model_layers_34_mlp_waleed_W_g/max_abs":2.46875,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/mean":-3.597233444452286e-08,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/norm":0.035160064490327506,"train/train/tensor_param_model_layers_52_mlp_waleed_W_u_weight/mean":-0.0002899169921875,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/max_abs":0.16015625,"train/train/tensor_act_model_layers_38_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_58/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp_down_proj/norm":378.593975088037,"train/train/tensor_act_model_layers_59_mlp_waleed_W_u/mean":0.002170562744140625,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_waleed_W_u_weight/max_abs":0.002288818359375,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn/mean":0.0005230903625488281,"train/train/tensor_param_model_layers_45_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_57_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/mean":1.531839370727539e-05,"train/train/tensor_grad_model_layers_22_mlp_waleed_W_u_weight/std":5.967530714236497e-05,"train/train/layer_model_layers_60/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_59_mlp_waleed_W_u/std":0.3730469430889385,"train/train/tensor_act_model_layers_48_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/std":2.245862689337389e-05,"train/train/tensor_act_model_layers_54_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/norm":3743.1841436604545,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_input_layernorm/norm":5792.606445312614,"train/train/layer__model_layers_63/param/mean":0.001551564137760823,"train/train/layer__model_layers_35/param/mean":0.001425899321129095,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/mean":1.1883676052093506e-06,"train/train/tensor_act_model_layers_55_self_attn_q_proj/std":0.9423843596253024,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/max_abs":0.1650390625,"train/train/tensor_act_model_layers_74_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_mlp_waleed_W_u_weight/mean":9.028735803440213e-08,"train/train/tensor_act_model_layers_42_mlp_waleed_W_u/mean":-0.003406524658203125,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/mean":7.073685992509127e-08,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/std":0.00010922360467352136,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_23_mlp/std":0.05310128793812479,"train/train/tensor_act_model_layers_50_mlp_waleed/std":0.1320807776651834,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/std":0.021484375,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn/max_abs":0.349609375,"train/train/tensor_grad_model_layers_12_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_v_proj/mean":0.00768280029296875,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_mlp_waleed_W_g_weight/norm":7.5625,"train/train/tensor_act_model_layers_13_mlp/std":0.09252962405123653,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/mean":0.0003185272216796875,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_waleed_W_g_weight/std":4.4230043519965736e-05,"train/train/tensor_act_model_layers_33_self_attn_v_proj/mean":0.00434112548828125,"train/train/tensor_param_model_layers_67_mlp_waleed_W_u_weight/std":0.032958984375,"train/train/tensor_act_model_layers_39_mlp_waleed_W_u/max_abs":2.59375,"train/train/layer__model_layers_46/param/mean":0.0014937902203588144,"train/train/tensor_act_model_layers_48_self_attn_k_proj/max_abs":6.1875,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_down_proj/norm":470.9768870425397,"train/train/tensor_grad_model_layers_42_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/max_abs":0.263671875,"train/train/tensor_act_model_layers_84_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/mean":-2.868473529815674e-07,"train/train/layer__model_layers_89/param/norm":25.852458478353658,"train/train/tensor_act_model_layers_72_mlp_waleed_W_u/mean":-0.0011053085327148438,"train/train/tensor_act_model_layers_80_self_attn_k_proj/norm":7285.088903248022,"train/train/tensor_act_model_layers_20_self_attn/std":0.10132224757214221,"train/train/tensor_act_model_layers_68_mlp/max_abs":0.90625,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/mean":-5.515175871551037e-08,"train/train/layer_model_layers_3/act/norm":23827.023832924056,"train/train/tensor_param_model_layers_90_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/mean":-4.9709342420101166e-08,"train/train/tensor_param_model_layers_69_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn/max_abs":5.5625,"train/train/tensor_act_model_layers_56/max_abs":24,"train/train/tensor_act_model_layers_89_mlp_waleed/norm":4506.680135951663,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_q_proj/mean":-0.01739501953125,"train/train/tensor_act_model_layers_72_mlp_waleed_W_u/std":0.4418954625664478,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/norm":0.0010680947983012913,"train/train/layer_model_layers_90/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_k_proj/mean":-0.003692626953125,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/std":0.0361328125,"train/train/tensor_act_model_layers_10_self_attn_k_proj/mean":-0.0058746337890625,"train/train/tensor_act_model_layers_82_self_attn_q_proj/max_abs":5.9375,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn_q_proj/std":1.1718750762939427,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/std":2.475613323689828e-05,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/mean":0.0002880096435546875,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/mean":3.9071892388165e-09,"train/train/tensor_grad_model_layers_27_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_86_mlp_waleed_W_g_weight/norm":8.25,"train/train/tensor_act_model_layers_72_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp_waleed_W_u/std":0.26416156516342826,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/norm":5.78125,"train/train/tensor_param_model_layers_77_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_45_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/max_abs":0.00022792816162109375,"train/train/tensor_act_model_layers_61_self_attn/mean":-5.124509334564209e-05,"train/train/layer__model_layers_5/param/std":0.04757727682762183,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/max_abs":0.1953125,"train/train/layer__model_layers_91/param/mean":0.0015596577232229914,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_52/param/std":0.0493421534152406,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/norm":5.40625,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/mean":5.3638359531760216e-08,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/mean":4.410743713378906e-05,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/std":0.00014167897430884314,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/mean":5.066394805908203e-06,"train/train/layer_model_layers_72/act/norm":21256.988538019337,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_v_proj/max_abs":2.359375,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/std":5.5804219286390996e-05,"train/train/layer_model_layers_38/act/max_abs":25.125,"train/train/tensor_act_model_layers_56_post_attention_layernorm/max_abs":5.03125,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_61_mlp_waleed_W_g/mean":0.0080108642578125,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/norm":5.125,"train/train/tensor_param_model_layers_9_mlp_waleed_W_u_weight/mean":-7.2479248046875e-05,"train/train/layer__model_layers_20/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_mlp_waleed_W_g_weight/max_abs":0.201171875,"train/train/tensor_act_model_layers_57_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/std":0.0264892578125,"train/train/tensor_act_model_layers_56_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_q_proj/max_abs":4.1875,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/std":0.031494140625,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/std":0.00018294035618488742,"train/train/tensor_act_model_layers_16_input_layernorm/mean":-0.00534820556640625,"train/train/tensor_act_model_layers_20_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_59/grad/max_abs":0.00118255615234375,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_input_layernorm/mean":0.0003432035446166992,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/norm":0.0015678498576462653,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/std":3.1150919936560656e-05,"train/train/tensor_act_model_layers_76_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_mlp_waleed_W_u_weight/std":0.0277099609375,"train/train/tensor_act_model_layers_9_mlp_waleed_W_g/mean":0.0028076171875,"train/train/tensor_act_model_layers_6_mlp_waleed/norm":2578.3462082925585,"train/train/tensor_act_model_layers_74_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_mlp_waleed_W_g_weight/norm":5.84375,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_37/param/std":0.04896175951741786,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/mean":-0.00019741058349609375,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/mean":2.4557113647460938e-05,"train/train/tensor_act_model_layers_36_mlp_waleed_W_u/norm":2328.5779404038017,"train/train/tensor_act_model_layers_76_self_attn_v_proj/max_abs":4.5625,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/norm":0.003538443764963443,"train/train/tensor_grad_model_layers_53_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_2/act/max_abs":27.5,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/std":5.2937965450090795e-05,"train/train/tensor_param_model_layers_2_mlp_waleed_W_u_weight/std":0.0244140625,"train/train/tensor_act_model_layers_70_mlp_waleed/max_abs":4,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_82/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_91/act/frac_near_dtype_limit":0,"train/train/layer__model_layers_44/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/norm":4.9375,"train/train/tensor_act_model_layers_77_mlp_down_proj/max_abs":1.5234375,"train/train/layer__model_layers_19/param/mean":0.0016735630362714508,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/norm":4.5625,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/std":0.0537109375,"train_steps_per_second":0.389,"train/train/tensor_act_model_layers_36_self_attn/max_abs":3.78125,"train/train/tensor_act_model_layers_53_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_mlp/max_abs":0.90234375,"train/train/tensor_act_model_layers_36_self_attn_k_proj/std":0.9746113787665043,"train/train/tensor_grad_model_layers_51_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/std":8.803454327932369e-05,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/norm":0.003635796749147202,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_input_layernorm/norm":5792.614990235748,"train/train/tensor_act_model_layers_33_mlp_down_proj/max_abs":0.337890625,"train/train/tensor_grad_model_layers_45_mlp_waleed_W_g_weight/norm":0.013138340176449454,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_k_proj/mean":0.0899658203125,"train/train/tensor_act_model_layers_10_mlp_down_proj/mean":-0.0006723403930664062,"train/train/tensor_act_model_layers_80_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/norm":4.78125,"train/train/tensor_act_model_layers_19_input_layernorm/norm":5792.60937500569,"train/train/tensor_act_model_layers_91_self_attn_v_proj/std":0.6806662154471502,"train/train/tensor_act_model_layers_31_self_attn_o_proj/std":0.15063544206035434,"train/train/tensor_act_model_layers_20_self_attn_o_proj/norm":587.135952066428,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/norm":4.03125,"train/train/tensor_act_model_layers_41_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/mean":-0.0001068115234375,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/std":0.035400390625,"train/train/tensor_grad_model_layers_87_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn/max_abs":1.3671875,"train/train/tensor_act_model_layers_81_mlp/mean":0.00701141357421875,"train/train/tensor_act_model_layers_45_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_down_proj/max_abs":0.91015625,"train/train/tensor_param_model_layers_92_mlp_waleed_W_g_weight/norm":10.6875,"train/train/tensor_act_model_layers_15_post_attention_layernorm/std":1.000000013358658,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_27_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/std":0.0272216796875,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85/max_abs":30,"train/train/tensor_act_model_layers_10_post_attention_layernorm/mean":-0.005615234375,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/norm":0.0011283123927658077,"train/train/tensor_act_model_layers_93_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_waleed_W_u_weight/max_abs":0.00156402587890625,"train/train/tensor_act_model_layers_77_self_attn_v_proj/max_abs":4.125,"train/train/tensor_act_model_layers_32_input_layernorm/mean":0.0014071464538574219,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/std":0.02880859375,"train/train/tensor_act_model_layers_15_input_layernorm/norm":5792.6123046885805,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/max_abs":0.000644683837890625,"train/train/tensor_act_model_layers_2_mlp_waleed_W_u/max_abs":2.890625,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_2/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_o_proj/mean":-0.0026726722717285156,"train/train/tensor_act_model_layers_11_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/mean":-1.2945383787155151e-07,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/std":0.0284423828125,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/mean":6.103515625e-05,"train/train/tensor_grad_model_layers_68_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn/norm":529.0076487260571,"train/train/tensor_act_model_layers_89_mlp_waleed/std":0.5507812654692017,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_k_proj/norm":7183.5276767971345,"train/train/tensor_param_model_layers_49_mlp_waleed_W_u_weight/mean":-6.67572021484375e-05,"train/train/tensor_param_model_layers_93_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/norm":4.75,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/mean":5.340576171875e-05,"train/train/tensor_act_model_layers_74_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_mlp_waleed/norm":901.9283915573159,"train/train/tensor_param_model_layers_85_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_76_mlp_waleed_W_g_weight/std":6.926838550835719e-05,"train/train/tensor_act_model_layers_79_self_attn_o_proj/max_abs":3.78125,"train/train/layer_model_layers_29/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp_waleed_W_u/std":0.4995128705807566,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/max_abs":0.000522613525390625,"train/train/tensor_act_model_layers_40/mean":0.005451202392578125,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74/max_abs":25.875,"train/train/tensor_act_model_layers_92/max_abs":35.75,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/max_abs":0.271484375,"train/train/layer__model_layers_68/param/mean":0.0014988368089411076,"train/train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_24/param/mean":0.0015541588460413417,"train/train/tensor_act_model_layers_17_self_attn_o_proj/std":0.18408401210564038,"train/train/tensor_param_model_layers_5_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/norm":0.014694301014737845,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/max_abs":0.0003509521484375,"train/train/tensor_act_model_layers_47_post_attention_layernorm/mean":0.005992889404296875,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/mean":8.392333984375e-05,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp_waleed_W_g/mean":-0.0142364501953125,"train/train/tensor_act_model_layers_18_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_16/act/norm":21365.667588652803,"train/train/tensor_act_model_layers_67_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/mean":-1.4217221178114414e-07,"train/train/tensor_param_model_layers_68_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_92/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_mlp_waleed_W_g_weight/mean":2.6775524020195007e-08,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp_down_proj/norm":352.1878185351718,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/std":0.0478515625,"train/train/tensor_param_model_layers_7_mlp_waleed_W_g_weight/mean":-0.000293731689453125,"train/train/tensor_param_model_norm_weight/std":0,"train/train/layer_model_layers_5/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/norm":0.034691601072627073,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/mean":1.3094177120365202e-07,"train/train/layer_model_layers_60/grad/max_abs":0.00098419189453125,"train/train/tensor_param_model_layers_1_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/std":0.04736328125,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/mean":2.4330802261829376e-07,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/mean":2.852175384759903e-08,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/norm":2.71875,"train/train/tensor_param_model_layers_30_mlp_waleed_W_u_weight/norm":4.46875,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/max_abs":0.20703125,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/mean":0.000659942626953125,"train/train/tensor_act_model_layers_18/mean":-0.00858306884765625,"train/train/tensor_param_model_layers_4_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_waleed_W_g_weight/norm":0.6383492591139495,"train/train/tensor_act_model_layers_41_self_attn_k_proj/norm":4650.109436149786,"train/train/layer_model_layers_17/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_26_mlp_waleed_W_g/norm":2223.4154844081536,"train/train/layer_model_layers_65/act/norm":19916.897682688592,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn_o_proj/std":0.604501622485357,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn_v_proj/max_abs":2.171875,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_1_mlp_waleed_W_g_weight/mean":0.00030517578125,"train/train/tensor_act_model_layers_82_self_attn_k_proj/mean":0.096435546875,"train/train/tensor_grad_model_layers_93_mlp_waleed_W_g_weight/max_abs":0.00148773193359375,"train/train/layer__model_layers_3/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/std":0.0252685546875,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/mean":1.282314769923687e-07,"train/train/tensor_act_model_layers_73_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_o_proj/norm":3502.8440548790777,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/std":0.034912109375,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/mean":7.180497050285339e-07,"train/train/tensor_act_model_layers_38_mlp_down_proj/mean":-5.8025121688842773e-05,"train/train/tensor_act_model_layers_52_input_layernorm/norm":5792.606933596662,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_self_attn/mean":-2.7298927307128906e-05,"train/train/tensor_act_model_layers_52_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57/mean":0.007404327392578125,"train/train/tensor_act_model_layers_76/mean":0.0126800537109375,"train/train/layer_model_layers_9/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_post_attention_layernorm/std":1.0000011826043476,"train/train/tensor_act_model_layers_11_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_55/act/norm":19276.18525417815,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/std":0.038818359375,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_20_mlp_down_proj/max_abs":0.53515625,"train/train/tensor_act_model_layers_85_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_48/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/std":0.0267333984375,"train/train/tensor_act_model_layers_24_self_attn/std":0.21997220454830863,"train/train/tensor_act_model_layers_5_post_attention_layernorm/std":1.0000001048901994,"train/train/tensor_act_model_layers_76_mlp_waleed/norm":1927.5004962151095,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp/std":0.05383349134977785,"train/train/tensor_act_model_layers_49_mlp_waleed/mean":0.0014095306396484375,"train/train/tensor_act_model_layers_70_self_attn_v_proj/max_abs":3.421875,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_43/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_5/grad/mean":-1.044452478077594e-07,"train/train/tensor_param_model_layers_31_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_input_layernorm/std":1.0000001189764518,"train/train/tensor_grad_model_layers_49_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_waleed_W_u_weight/mean":-7.031485438346863e-07,"train/train/tensor_act_model_layers_9_self_attn_q_proj/mean":-0.03472900390625,"train/train/tensor_act_model/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn_o_proj/std":0.07007228272442838,"train/train/tensor_param_model_layers_56_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/max_abs":0.0014801025390625,"train/train/tensor_grad_model_layers_80_mlp_waleed_W_g_weight/mean":3.878958523273468e-07,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/mean":0.00014019012451171875,"train/train/tensor_act_model_layers_73_self_attn_o_proj/mean":-0.004009246826171875,"train/train/tensor_param_model_layers_87_mlp_waleed_W_g_weight/norm":8.625,"train/train/tensor_act_model_layers_53_mlp_waleed/mean":-0.0008029937744140625,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/norm":0.03912824819566904,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/max_abs":0.0002651214599609375,"train/train/tensor_param_model_layers_38_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_5_self_attn_o_proj/std":0.22241252775210488,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/norm":0.028903288302897002,"train/train/tensor_grad_model_layers_89_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/std":0.00010089062332668782,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_down_proj/max_abs":0.439453125,"train/train/tensor_act_model_layers_33_self_attn_o_proj/mean":9.223818778991699e-05,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/norm":0.012938769340441716,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp/max_abs":0.62109375,"train/train/tensor_param_model_layers_66_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_12_self_attn_k_proj/norm":6503.380423609588,"train/train/tensor_act_model_layers_7_input_layernorm/std":1.000000090105455,"train/train/tensor_grad_model_layers_91_mlp_waleed_W_g_weight/std":0.00012440906791021626,"train/train/tensor_grad_model_layers_10_mlp_waleed_W_u_weight/mean":-2.859160304069519e-07,"train/train/tensor_param_model_layers_88_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_10_self_attn_o_proj/std":0.16870317250845782,"train/train/tensor_act_model_layers_1_mlp_waleed_W_g/norm":7149.818573888306,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/norm":0.017619296706785054,"train/train/tensor_act_model_layers_29_input_layernorm/std":1.0000002742176173,"train/train/tensor_act_model_layers_35_mlp_waleed_W_g/max_abs":2.671875,"train/train/tensor_act_model_layers_24_self_attn_k_proj/std":1.1093752199495124,"train/train/layer__model_layers_9/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_34/param/norm":19.63453316425234,"train/train/tensor_param_model_layers_46_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/norm":4.375,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/std":4.119919444686609e-05,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/std":0.00010837776903770598,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/norm":0.0006471788779467921,"train/train/tensor_act_model_layers_70/mean":0.01113128662109375,"train/train/layer_model_layers_78/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/norm":6.28125,"train/train/layer__model_layers_2/param/norm":18.979944184586003,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/norm":5.09375,"train/train/tensor_act_model_layers_26_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/mean":6.400048732757568e-06,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/max_abs":0.0002117156982421875,"train/train/tensor_param_model_layers_33_mlp_waleed_W_u_weight/norm":4.46875,"train/train/tensor_param_model_layers_43_mlp_waleed_W_u_weight/mean":-8.58306884765625e-05,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/std":8.501461397952655e-05,"train/train/tensor_act_model_layers_35_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/std":0.0001353218425851378,"train/train/tensor_act_model_layers_82_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_39/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_o_proj/mean":0.0156402587890625,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/max_abs":0.000522613525390625,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/std":0.036376953125,"train/train/tensor_act_model_layers_12_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_mlp_waleed_W_g_weight/max_abs":0.150390625,"train/train/tensor_act_model_layers_92_mlp_down_proj/norm":7902.898941645324,"train/train/tensor_param_model_layers_82_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_waleed_W_g/std":0.3085938868454437,"train/train/tensor_act_model_layers_22_self_attn/std":0.10779024619608649,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/mean":-1.0468284017406404e-08,"train/train/tensor_act_model_layers_81_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/norm":0.02913455501060109,"train/train/tensor_grad_model_layers_0_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/std":0.0238037109375,"train/train/tensor_param_model_layers_12_mlp_waleed_W_u_weight/max_abs":0.10498046875,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/mean":2.1457672119140625e-05,"train/train/tensor_act_model_layers_11_post_attention_layernorm/mean":-0.00580596923828125,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp_waleed_W_u/norm":4082.514879033183,"train/train/tensor_act_model_layers_26_mlp_waleed_W_g/max_abs":2.015625,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/std":0.047119140625,"train/train/tensor_act_model_layers_74_self_attn_k_proj/mean":-0.02374267578125,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/mean":6.294430932030082e-08,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/mean":1.1491647455841303e-07,"train/train/tensor_act_model_layers_23_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/std":7.917890108773446e-05,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/std":0.0225830078125,"train/train/layer_model_layers_89/grad/mean":7.113793431875672e-08,"train/train/tensor_act_model_layers_92_self_attn_k_proj/mean":-0.0347900390625,"train/train/tensor_act_model_layers_7_self_attn_k_proj/mean":-0.00791168212890625,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/std":4.244019299173716e-05,"train/train/tensor_act_model_layers_82_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/max_abs":0.26171875,"train/train/tensor_act_model_layers_82/norm":18897.901448553053,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_61_post_attention_layernorm/mean":0.00414276123046875,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/std":4.1784025939475155e-05,"train/train/tensor_act_/frac_near_dtype_limit":0,"train/train/layer_model_layers_46/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/mean":-2.0081643015146255e-08,"train/train/tensor_act_model_layers_86_post_attention_layernorm/std":1.0000001268721155,"train/train/layer__model_layers_76/param/mean":0.0015533375851635627,"train/train/tensor_act_model_layers_7_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_mlp_waleed_W_u_weight/max_abs":0.000598907470703125,"train/train/tensor_act_model_layers_5_self_attn_v_proj/std":0.39648453472867107,"train/train/tensor_act_model_layers_46_mlp_down_proj/max_abs":0.8515625,"train/train/layer__model_layers_1/param/norm":19.56936905161035,"train/train/layer__model_layers_76/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/max_abs":0.1259765625,"train/train/tensor_grad_model_norm_weight/std":0.002167097251583412,"train/train/layer_model_layers_40/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/mean":8.487701416015625e-05,"train/train/tensor_act_model_layers_83_mlp_waleed_W_g/max_abs":3.453125,"train/train/tensor_act_model_layers_1_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/norm":4.375,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/mean":7.915496826171875e-05,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/norm":5.8125,"train/train/tensor_act_model_layers_88_self_attn_o_proj/max_abs":3.984375,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/std":1.899651583602365e-05,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/max_abs":0.22265625,"train/train/tensor_act_model_layers_13_mlp_waleed/mean":0.0005426406860351562,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/mean":0.00014019012451171875,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/norm":5.21875,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/std":0.040771484375,"train/train/tensor_act_model_layers_56_mlp_down_proj/std":0.08300810673578045,"train/train/tensor_act_model_layers_71_mlp_down_proj/mean":-0.0006818771362304688,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/norm":0.005516574565803162,"train/train/tensor_act_model_layers_49_self_attn_k_proj/std":0.7275410568926823,"train/train/tensor_act_model_layers_18_self_attn/mean":0.00010704994201660156,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/norm":0.057095118255172064,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/max_abs":0.125,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/std":0.028564453125,"train/train/tensor_act_model_layers_82_self_attn/mean":0.0014188289642333984,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/std":0,"train/train/layer__model_layers_26/param/max_abs":1,"train/train/tensor_act_model_layers_37_self_attn_q_proj/norm":5591.866066354584,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_84_mlp_waleed_W_u/std":0.6582064812458622,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/norm":2.75,"train/train/tensor_act_model_layers_10_mlp_waleed/mean":0.008331298828125,"train/train/tensor_grad_model_layers_15_mlp_waleed_W_u_weight/norm":0.020219317692093553,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/norm":0.0008774591493484333,"train/train/tensor_act_model_layers_37_post_attention_layernorm/max_abs":5.375,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/norm":5.6875,"train/train/tensor_param_model_layers_17_mlp_waleed_W_g_weight/std":0.0234375,"train/train/tensor_act_model_layers_46_self_attn_q_proj/norm":5219.533822312615,"train/train/tensor_act_model_layers_69_input_layernorm/mean":0.005352020263671875,"train/train/tensor_act_model_layers_39/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/max_abs":0.2060546875,"train/train/tensor_act_model_layers_14_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_mlp_waleed_W_g_weight/mean":-5.054473876953125e-05,"train/train/tensor_act_model_layers_29_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp_waleed/mean":0.00311279296875,"train/train/layer_model_layers_41/grad/norm":0.029568361992170975,"train/train/tensor_act_model_layers_0_self_attn_k_proj/mean":-0.0114898681640625,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_52_self_attn_v_proj/max_abs":1.875,"train/train/tensor_act_model_layers_24_self_attn_k_proj/norm":6443.671057195432,"train/train/tensor_grad_model_layers_81_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_mlp_waleed_W_g/mean":0.00408935546875,"train/train/tensor_act_model_layers_66_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_12/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_51/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/mean":-7.33125489205122e-08,"train/train/tensor_param_model_layers_89_mlp_waleed_W_u_weight/norm":9.25,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/max_abs":0.20703125,"train/train/layer_model_layers_87/act/norm":27307.550025352044,"train/train/tensor_act_model_layers_82_post_attention_layernorm/max_abs":5.3125,"train/train/tensor_act_model_layers_38_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_46_mlp_waleed_W_u/mean":0.0067291259765625,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/max_abs":0.1025390625,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/norm":2.953125,"train/train/tensor_act_model_layers_49_input_layernorm/norm":5792.610717775591,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/std":7.989478629584075e-05,"train/train/tensor_act_model_layers_83_post_attention_layernorm/max_abs":5.28125,"train/train/tensor_param_model_layers_0_mlp_waleed_W_g_weight/max_abs":0.1455078125,"train/train/tensor_act_model_layers_42/mean":0.00832366943359375,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_65/grad/std":5.255274416285749e-05,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/mean":9.870529174804688e-05,"train/train/tensor_param_model_layers_52_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp/std":0.18725634005825564,"train/train/tensor_act_model_layers_51_mlp_waleed_W_g/norm":2898.443480207433,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_mlp_waleed_W_u_weight/mean":-7.82012939453125e-05,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn_v_proj/norm":2662.5068319120255,"train/train/tensor_act_model_layers_16_self_attn_v_proj/norm":2115.546094814383,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/std":0.030029296875,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/norm":4.28125,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/max_abs":0.0010223388671875,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_41/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp_waleed_W_g/max_abs":2.609375,"train/train/tensor_act_model_layers_52_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_q_proj/std":0.9951187672850523,"train/train/tensor_act_model_layers_57/max_abs":24,"train/train/layer__model_layers_15/param/max_abs":1,"train/train/tensor_act_model_layers_25_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/max_abs":0.2265625,"train/train/tensor_act_model_layers_56_mlp_waleed_W_u/max_abs":2.953125,"train/train/layer__model_layers_7/param/norm":19.25044388692765,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/mean":-0.0002002716064453125,"train/train/tensor_act_model_layers_83_mlp_waleed/std":0.3203125193667075,"train/train/tensor_act_model_layers_56_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_q_proj/std":1.1054770735572206,"train/train/tensor_grad_model_layers_74_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"_runtime":3875,"train/train/tensor_param_model_layers_8_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_0_self_attn_v_proj/max_abs":3.140625,"train/train/tensor_act_model_layers_45_input_layernorm/norm":5792.61059570524,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp/norm":345.7169320330921,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/std":8.888736143556272e-05,"train/train/layer_model_layers_7/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/std":4.424416148331195e-05,"train/train/tensor_act_model_layers_82_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_42_self_attn_q_proj/std":0.9814509498673892,"train/train/tensor_act_model_layers_45_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/std":9.03616398003945e-05,"train/train/tensor_act_model_layers_7_self_attn_k_proj/max_abs":7.4375,"train/train/layer_model_layers_52/act/norm":19125.742045358773,"train/train/layer_model_layers_6/act/mean":-0.0036484375596046448,"train/train/tensor_act_model_layers_8_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/norm":4.75,"train/train/tensor_grad_model_layers_53_mlp_waleed_W_u_weight/norm":0.01503061205765248,"train/train/tensor_act_model_layers_17_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/std":5.2157767767972586e-05,"train/train/tensor_act_model_layers_85/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn_v_proj/norm":2742.2017323250852,"train/train/tensor_param_model_layers_58_mlp_waleed_W_g_weight/norm":5.28125,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/norm":10.4375,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/mean":-0.0001430511474609375,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/std":0.021728515625,"train/train/tensor_act_model_layers_32_mlp_waleed_W_g/max_abs":2.125,"train/train/tensor_param_model_layers_17_mlp_waleed_W_u_weight/norm":4.28125,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/mean":0.00012683868408203125,"train/train/tensor_param_model_layers_15_mlp_waleed_W_g_weight/mean":0.000141143798828125,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/mean":-5.122274160385132e-07,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/max_abs":0.140625,"train/train/tensor_act_model_layers_31_mlp/std":0.06091370545236591,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/mean":5.054473876953125e-05,"train/train/tensor_param_model_layers_74_mlp_waleed_W_u_weight/mean":-7.963180541992188e-05,"train/train/tensor_act_model_layers_37_mlp_waleed/std":0.09863282843391366,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_mlp_waleed_W_g/norm":2945.7673864674425,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/norm":0.0037704804317465236,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/std":0.044921875,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_91_mlp_waleed_W_u_weight/max_abs":0.279296875,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/std":5.8419522881989826e-05,"train/train/tensor_act_model_layers_65_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_k_proj/std":1.5078125938850333,"train/train/tensor_act_model_layers_54_mlp_waleed/std":0.12255908815884509,"train/train/tensor_act_model_layers_5_self_attn_k_proj/norm":9570.239789004494,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/norm":0.005678074047424764,"train/train/tensor_act_model_layers_83/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/mean":0.00016307830810546875,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_norm_weight/max_abs":0.011474609375,"train/train/tensor_act_model_layers_91_mlp/std":1.042976535074239,"train/train/layer_model_layers_89/act/std":1.203112473397819,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/norm":4.9375,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/mean":-9.140931069850922e-07,"train/train/tensor_act_model_layers_83_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_mlp_waleed_W_g_weight/max_abs":0.22265625,"train/train/tensor_act_model_layers_9_mlp/max_abs":0.69140625,"train/train/tensor_act_model_layers_63_mlp_down_proj/std":0.1013187056580947,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54/norm":15080.385949893614,"train/train/tensor_grad_model_layers_80_mlp_waleed_W_g_weight/max_abs":0.000766754150390625,"train/train/tensor_grad_model_layers_72_mlp_waleed_W_g_weight/mean":2.2212043404579163e-07,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/max_abs":0.0003299713134765625,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_mlp_waleed_W_g/std":0.31250002309679903,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_85_post_attention_layernorm/norm":5792.611694339244,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_post_attention_layernorm/std":1.0000012022421336,"train/train/tensor_param_model_layers_48_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/max_abs":0.0001926422119140625,"train/train/tensor_act_model_layers_67_mlp_waleed_W_g/std":0.4316406331079847,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_1_mlp_waleed_W_g_weight/std":0.02880859375,"train/train/tensor_act_model_layers_42_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/mean":-9.19681042432785e-09,"train/train/tensor_act_model_layers_29/norm":15624.421941631295,"train/train/tensor_grad_model_layers_58_mlp_waleed_W_g_weight/norm":0.015262084605057335,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/mean":2.5358167476952076e-07,"train/train/layer_model_layers_5/act/std":1.065978917857083,"train/train/tensor_grad_model_layers_85_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/max_abs":0.00095367431640625,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/max_abs":0.2890625,"train/train/tensor_act_model_layers_77_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn/norm":297.90934763965004,"train/train/layer_model_layers_68/act/norm":20065.77571945325,"train/train/tensor_act_model_layers_83_self_attn_q_proj/std":1.2363337837559794,"train/train/tensor_param_model_layers_17_mlp_waleed_W_u_weight/std":0.0235595703125,"train/train/tensor_act_model_layers_39_mlp_waleed_W_u/std":0.29541150676008754,"train/train/tensor_act_model_layers_29_self_attn_k_proj/norm":5472.328651413215,"train/train/tensor_param_model_layers_82_mlp_waleed_W_u_weight/mean":-9.107589721679688e-05,"train/train/layer_model_layers_67/grad/max_abs":0.0010833740234375,"train/train/tensor_param_model_layers_88_mlp_waleed_W_g_weight/mean":-9.393692016601562e-05,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/max_abs":0.0005340576171875,"train/train/tensor_param_model_layers_74_mlp_waleed_W_g_weight/max_abs":0.19140625,"train/train/tensor_act_model_layers_48_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn/max_abs":1.34375,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/std":0.036865234375,"train/train/layer_model_layers_62/act/std":0.8405644735827277,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/mean":1.1827796697616577e-07,"train/train/tensor_act_model_layers_64_self_attn_k_proj/norm":5464.840632241888,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/norm":0.013283877954223049,"train/train/tensor_act_model_layers_51_mlp_waleed_W_g/std":0.3535157655319177,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/mean":6.193295121192932e-08,"train/train/tensor_param_model_layers_77_mlp_waleed_W_g_weight/norm":6.71875,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/norm":0.05078770523123711,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_23_mlp_waleed_W_u_weight/mean":-0.00011539459228515625,"train/train/layer_model_layers_55/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/mean":-0.000324249267578125,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/mean":2.7837813831865788e-08,"train/train/tensor_param_model_layers_13_mlp_waleed_W_g_weight/max_abs":0.15625,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/std":0.040283203125,"train/train/tensor_act_model_layers_78_mlp_waleed/std":0.24584999058390117,"train/train/tensor_grad_model_layers_79_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_o_proj/mean":-5.124509334564209e-05,"train/train/tensor_param_model_layers_80_mlp_waleed_W_u_weight/norm":7.09375,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/norm":0.0007748837100698917,"train/train/tensor_act_model_layers_34_mlp_down_proj/norm":345.7169320330921,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/mean":0.0001239776611328125,"train/train/tensor_act_model_layers_39_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_waleed_W_u/mean":-0.00439453125,"train/train/tensor_act_model_layers_0_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn/norm":872.910595406355,"train/train/tensor_act_model_layers_32_self_attn_q_proj/mean":0.0008563995361328125,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp/norm":389.85733023765897,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/norm":0.013672351828843111,"train/train/tensor_act_model_layers_25_mlp/max_abs":0.66796875,"train/train/tensor_act_model_layers_54_mlp_waleed_W_g/max_abs":2.6875,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/mean":1.2619420886039734e-07,"train/train/tensor_act_model_layers_54_self_attn_q_proj/norm":5190.214011337085,"train/train/tensor_grad_model_layers_59_mlp_waleed_W_g_weight/mean":-9.522773325443268e-08,"train/train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_mlp_waleed_W_u/norm":3237.205090971687,"train/train/layer_model_layers_66/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn/norm":499.8525820415348,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/mean":5.1021575927734375e-05,"train/train/tensor_act_model_layers_10_self_attn_o_proj/max_abs":1.28125,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_81/grad/max_abs":0.001251220703125,"train/train/layer_model_layers_30/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn/max_abs":1.9296875,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/norm":5.09375,"train/train/tensor_act_model_layers_19_self_attn_k_proj/max_abs":4.71875,"train/train/tensor_act_model_layers_9_mlp_waleed_W_g/std":0.2753906269022759,"train/train/tensor_act_model_layers_67_post_attention_layernorm/norm":5792.611572267564,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/max_abs":0.000713348388671875,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/std":0.025634765625,"train/train/layer_model_layers_66/grad/std":5.452007597217482e-05,"train/train/tensor_act_model_layers_80_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_mlp_waleed/max_abs":2.578125,"train/train/tensor_act_model_layers_8/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15/mean":-0.006256103515625,"train/train/tensor_param_model_layers_82_mlp_waleed_W_u_weight/max_abs":0.197265625,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/max_abs":0.000537872314453125,"train/train/tensor_act_model_layers_23_self_attn_o_proj/mean":0.0105743408203125,"train/train/layer_model_layers_10/grad/norm":0.07689639081128355,"train/train/tensor_param_model_layers_32_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_4_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_o_proj/max_abs":4.21875,"train/train/tensor_act_model_layers_19_post_attention_layernorm/std":1.0000000154541338,"train/train/tensor_act_model_layers_50_mlp_down_proj/mean":0.002796173095703125,"train/train/tensor_act_model_layers_78/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/norm":0.013553034766848744,"train/train/tensor_act_model_layers_83_post_attention_layernorm/mean":0.009002685546875,"train/train/tensor_act_model_layers_16_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/norm":0.013902456805634207,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/mean":-6.4849853515625e-05,"train/train/tensor_act_model_layers_11_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/mean":6.489455699920654e-06,"train/train/tensor_act_model_layers_90_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_waleed_W_u_weight/std":0.00012544541576901672,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/norm":6.6875,"train/train/tensor_act_model_layers_85_self_attn_k_proj/mean":0.24267578125,"train/train/tensor_act_model_layers_73_mlp_waleed_W_g/std":0.453614084019274,"train/train/layer_model_layers_91/grad/norm":0.0858875872737072,"train/train/tensor_act_model_layers_16_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/norm":0.0018117201767685945,"train/train/tensor_act_model_layers_67/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_0_mlp_waleed_W_g/max_abs":5.96875,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_mlp_waleed_W_g_weight/norm":0.01880180038140616,"train/train/tensor_act_model_layers_58/max_abs":23.875,"train/train/tensor_param_model_layers_6_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_waleed_W_g_weight/max_abs":0.1953125,"train/train/tensor_act_model_layers_37_self_attn_o_proj/std":0.10876771754840547,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/std":8.637828474798807e-05,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/mean":-2.2619962692260742e-05,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/mean":-0.00011587142944335938,"train/train/tensor_act_model_layers_92_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/std":1.1021142682751155e-05,"train/train/tensor_param_model_layers_55_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_69_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_mlp_waleed/std":0.10302735115604095,"train/train/tensor_act_model_layers_26_self_attn_o_proj/norm":440.58534308160813,"train/train/tensor_act_model_layers_89_post_attention_layernorm/max_abs":5.90625,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/std":6.0456792375918425e-05,"train/train/tensor_act_model_layers_82_post_attention_layernorm/std":1.0000004358588699,"train/train/tensor_grad_model_layers_37_mlp_waleed_W_g_weight/std":4.2320816955108476e-05,"train/train/tensor_grad_model_layers_72_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/mean":2.1219253540039062e-05,"train/train/tensor_act_model_layers_76_mlp_waleed_W_u/norm":4090.9491224761096,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/max_abs":0.001434326171875,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92/norm":28678.390588912058,"train/train/tensor_act_model_layers_34_input_layernorm/mean":0.001804351806640625,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_input_layernorm/norm":5792.270996097208,"train/train/tensor_act_model_layers_66_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/mean":8.562346920371056e-08,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/norm":3.265625,"train/train/tensor_act_model_layers_69_self_attn_o_proj/norm":1280.2571859913237,"train/train/tensor_act_model_layers_78_mlp_waleed_W_g/norm":4082.457223229689,"train/train/tensor_act_model_layers_69_mlp_waleed_W_u/mean":0.0065460205078125,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_89/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/max_abs":5.21875,"train/train/tensor_act_model_layers_24_mlp_waleed_W_g/mean":0.0120391845703125,"train/train/tensor_act_model_layers_10/norm":17523.465213794796,"train/train/layer__model_layers_93/param/norm":29.144591720634892,"train/train/tensor_grad_model_layers_23_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/norm":0.0016282815318450084,"train/train/tensor_act_model_layers_63_mlp/std":0.1013187056580947,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/max_abs":0.248046875,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp_down_proj/std":0.09863283671229484,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/max_abs":0.0002002716064453125,"train/train/tensor_act_model_layers_9/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn/norm":1128.4717260247437,"train/train/tensor_act_model_layers_35_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/norm":6.4375,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/std":4.8186094785152785e-05,"train/train/tensor_act_model_layers_59_self_attn/norm":997.9807777449652,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/norm":2.984375,"train/train/tensor_param_model_layers_64_mlp_waleed_W_u_weight/max_abs":0.1650390625,"train/train/tensor_act_model_layers_51_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/norm":0.011382442999859814,"train/train/layer_model_layers_12/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/norm":0.003903842720708014,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/max_abs":0.35546875,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/mean":-2.2951513528823853e-05,"train/train/tensor_act_model_layers_71_self_attn_o_proj/std":0.13647543129551903,"train/train/tensor_act_model_layers_69_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp_down_proj/mean":0.00251007080078125,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/std":0.0238037109375,"train/train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_85/param/std":0.06111964828589146,"train/train/layer_model_layers_47/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn/norm":850.5285257797606,"train/train/tensor_act_model_layers_25_mlp_waleed_W_u/mean":0.00211334228515625,"train/train/tensor_act_model_layers_9_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85/std":3.4336008653903356,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/norm":0.00975818350508869,"train/train/tensor_act_model_layers_36_mlp_waleed/norm":836.4283417731708,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_q_proj/norm":6101.72106965118,"train/train/tensor_act_model_layers_49_self_attn/max_abs":0.50390625,"train/train/tensor_act_model_layers_3_self_attn_q_proj/std":0.7851563963427336,"train/train/tensor_act_model_layers_29_self_attn_v_proj/std":0.4062500139698384,"train/train/tensor_act_model_layers_72_self_attn/max_abs":3,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78/std":2.9648525634642167,"train/train/layer_model_layers_19/act/std":0.8822855534116454,"train/train/layer__model_layers_1/param/std":0.048291668509047744,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/mean":-5.691312253475189e-06,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/std":1.4253739037416136e-05,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/std":0.042236328125,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_v_proj/mean":0.00499725341796875,"train/train/tensor_act_model_layers_21_mlp/mean":0.001857757568359375,"train/train/tensor_act_model_layers_28/norm":15675.514096369126,"train/train/tensor_act_model_layers_41_self_attn/norm":267.3777887339979,"train/train/tensor_grad_model_layers_13_mlp_waleed_W_g_weight/max_abs":0.00372314453125,"train/train/tensor_act_model_layers_32_mlp_waleed_W_u/std":0.29248185216033407,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/std":5.941539057409981e-05,"train/train/tensor_act_model_layers_3_self_attn_q_proj/mean":-0.0067901611328125,"train/train/tensor_act_model_layers_86_self_attn_k_proj/norm":6017.069985945932,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/max_abs":0.1416015625,"train/train/tensor_act_model_layers_80_mlp_waleed/std":0.3027343903818434,"train/train/layer_model_layers_53/act/max_abs":23.625,"train/train/tensor_param_model_layers_79_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp/frac_near_user_limit":0,"train/train/layer__model_layers_54/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp/max_abs":0.66796875,"train/train/tensor_grad_model_layers_71_mlp_waleed_W_g_weight/norm":0.02023451212243709,"train/train/tensor_act_model_layers_7_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_50/act/norm":19605.40743451408,"train/train/tensor_act_model_layers_5_mlp_waleed/std":0.17749074138269852,"train/train/tensor_act_model_layers_0_mlp_down_proj/mean":-0.020477294921875,"train/train/tensor_act_model_layers_61/norm":14988.446058473,"train/train/layer__model_layers_89/param/std":0.06380057353583277,"train/train/tensor_act_model_layers_51_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_77/grad/mean":1.5460354291983587e-08,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/norm":4.3125,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/norm":11.625,"train/train/tensor_act_model_layers_62_mlp_down_proj/norm":547.7848352225518,"train/train/tensor_act_model_layers_63_self_attn/max_abs":1.7578125,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/norm":0.000609453852416454,"train/train/tensor_act_model_layers_59_self_attn_v_proj/norm":2338.2288077250514,"train/train/tensor_act_model_layers_28_self_attn_q_proj/mean":-0.04071044921875,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/norm":5.78125,"train/train/tensor_act_model_layers_82_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp_waleed_W_g/std":0.7080098584573058,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_0/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_mlp_waleed_W_g_weight/mean":0.00025177001953125,"train/train/tensor_param_model_layers_29_mlp_waleed_W_g_weight/max_abs":0.119140625,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/std":0.00013051780016664195,"train/train/tensor_grad_model_layers_20_mlp_waleed_W_g_weight/mean":-9.546056389808655e-08,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/max_abs":0.220703125,"train/train/layer_model_layers_60/act/max_abs":24.125,"train/train/tensor_grad_model_layers_85_mlp_waleed_W_g_weight/std":8.17990207568628e-05,"train/train/tensor_param_model_layers_33_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_mlp_waleed_W_u_weight/mean":9.16188582777977e-08,"train/train/tensor_act_model_layers_7/max_abs":26.875,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp/max_abs":0.671875,"train/train/tensor_act_model_layers_25/norm":15977.893168365514,"train/train/tensor_act_model_layers_75_mlp_waleed/max_abs":5.8125,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/std":2.2638633752664763e-05,"train/train/tensor_act_model_layers_4/std":3.3555121995184867,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/mean":9.350478649139404e-07,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/norm":4.75,"train/train/tensor_grad_model_layers_8_mlp_waleed_W_u_weight/norm":0.025181939767918642,"train/train/tensor_act_model_layers_3_mlp_waleed/max_abs":7.84375,"train/train/tensor_act_model_layers_25_mlp_waleed/max_abs":2.375,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/norm":0.02560895801458403,"train/train/tensor_act_model_layers_14_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_q_proj/norm":5777.011188877232,"train/train/layer__model_layers_50/param/std":0.049840287912691454,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/mean":-1.5343539416790009e-07,"train/train/tensor_act_model_layers_23_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp_down_proj/std":0.18725634005825564,"train/train/tensor_act_model_layers_50/mean":0.006656646728515625,"train/train/tensor_act_model_layers_88_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/std":0.046630859375,"train/train/tensor_act_model_layers_71/mean":0.01139068603515625,"train/train/tensor_act_model_layers_3_self_attn_k_proj/std":0.9453125266004196,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/mean":1.0652001947164536e-07,"train/train/tensor_act_model_layers_59_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_75/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/std":0.048583984375,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/mean":-0.0003948211669921875,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/mean":7.62939453125e-05,"train/train/tensor_act_model_layers_1_input_layernorm/norm":5792.601562504476,"train/train/layer_model_layers_43/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_mlp_waleed_W_g_weight/max_abs":0.00098419189453125,"train/train/tensor_act_model_layers_3_post_attention_layernorm/max_abs":4.46875,"train/train/tensor_act_model_layers_20/max_abs":26.125,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/norm":5.40625,"train/train/tensor_act_model_layers_40_self_attn_v_proj/max_abs":3.171875,"train/train/tensor_grad_model_layers_11_mlp_waleed_W_g_weight/norm":0.017090933628751786,"train/train/tensor_grad_model_layers_17_mlp_waleed_W_g_weight/max_abs":0.0009613037109375,"train/train/tensor_act_model_layers_37/norm":15379.322341661049,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32/max_abs":26.375,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/mean":-4.857778549194336e-05,"train/train/tensor_act_model_layers_22_input_layernorm/norm":5792.608276369291,"train/train/layer__model_layers_6/param/max_abs":1,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/norm":0.015243675376346577,"train/train/tensor_act_model_layers_46_input_layernorm/mean":0.006072998046875,"train/train/tensor_act_model_layers_31_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_77/param/mean":0.0017886823871392356,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/norm":5.40625,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/max_abs":0.228515625,"train/train/tensor_act_model_layers_38_post_attention_layernorm/std":1.0000008652923165,"train/train/tensor_act_model_layers_9_self_attn_v_proj/max_abs":2.28125,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/max_abs":0.251953125,"train/train/tensor_act_model_layers_16_mlp_waleed_W_g/max_abs":1.8671875,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/mean":1.9674189388751984e-07,"train/train/layer_model_layers_58/act/norm":20040.54130656291,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/norm":0.017519322694963896,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp_down_proj/std":0.281250030630163,"train/train/tensor_act_model_layers_20_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp_waleed_W_u/std":0.34375014901157963,"train/train/tensor_act_model_layers_62_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_46/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/norm":3.375,"train/train/tensor_act_model_layers_35_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41/mean":0.007472991943359375,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/max_abs":0.00011491775512695312,"train/train/tensor_param_model_layers_25_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/std":0.04248046875,"train/train/tensor_act_model_layers_14_mlp_waleed/std":0.10998556269550243,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_22_mlp/mean":0.0011615753173828125,"train/train/tensor_param_model_layers_8_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/mean":-4.553794860839844e-05,"train/train/tensor_act_model_layers_57_mlp_waleed/mean":-0.0009021759033203125,"train/train/layer_model_layers_24/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn/norm":5550.08046715998,"train/train/tensor_act_model_layers_7_mlp_waleed_W_g/std":0.3300781679179866,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/mean":-0.0003108978271484375,"train/train/tensor_act_model_layers_75_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_waleed_W_g_weight/std":4.2439521397251313e-05,"train/train/tensor_act_model_layers_88_mlp_waleed_W_u/mean":0.0090789794921875,"train/train/tensor_param_model_layers_77_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp_waleed_W_u/norm":1992.4557840486702,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/norm":0.0008597827514346952,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp/mean":-7.243454456329346e-05,"train/train/tensor_act_model_layers_37_input_layernorm/max_abs":5.46875,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/mean":-0.00016880035400390625,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/norm":5.3125,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/std":7.79240256277388e-05,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/std":0.00019062767358572622,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_83/act/norm":24741.676848692598,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_27_post_attention_layernorm/max_abs":5.59375,"train/train/tensor_act_model_layers_56_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_mlp_waleed_W_g_weight/norm":4.5625,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_q_proj/norm":5770.914238345273,"train/train/tensor_act_model_layers_85_mlp_down_proj/frac_near_user_limit":0,"eval/steps_per_second":4.396,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/std":0.0390625,"train/train/tensor_act_model_layers_89_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/norm":0.03535890450293108,"train/train/tensor_param_model_layers_51_mlp_waleed_W_g_weight/std":0.0279541015625,"train/train/tensor_act_model_layers_15_self_attn_q_proj/max_abs":8.3125,"train/train/layer_model_layers_51/act/frac_near_user_limit":0,"train/train/layer_model_layers_52/grad/norm":0.030046205856127488,"train/train/tensor_grad_model_layers_29_mlp_waleed_W_u_weight/mean":-8.477945812046528e-08,"train/train/tensor_act_model_layers_19_mlp_waleed_W_g/norm":2089.112861097231,"train/train/tensor_act_model_layers_73_mlp/max_abs":1.484375,"train/train/tensor_param_model_layers_2_input_layernorm_weight/mean":1,"train/train/layer__model_layers_65/param/max_abs":1,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_6/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/norm":0.0004912459625119819,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/max_abs":0.00032806396484375,"train/train/tensor_act_model_layers_40_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_waleed_W_u/std":0.503906377825812,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_73/grad/max_abs":0.0014801025390625,"train/train/tensor_grad_model_layers_56_mlp_waleed_W_g_weight/std":4.382739326220169e-05,"train/train/tensor_param_model_layers_11_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/mean":2.2659078240394592e-06,"train/train/tensor_act_model_layers_58_mlp_waleed/norm":1069.5263326382203,"train/train/tensor_act_model_layers_87_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_waleed_W_g_weight/max_abs":0.00051116943359375,"train/train/tensor_act_model_layers_47/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_post_attention_layernorm/norm":5792.609741212341,"train/train/tensor_act_model_layers_31_self_attn/std":0.15063544206035434,"train/train/tensor_act_model_layers_19_self_attn/norm":365.7470157363681,"train/train/tensor_act_model_layers_42_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_q_proj/max_abs":5.875,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/std":0.0299072265625,"train/train/layer_model_layers_50/act/mean":0.004938840866088867,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/std":0.02880859375,"train/train/layer__model_layers_76/param/norm":22.65915929596683,"train/train/tensor_param_model_layers_55_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_65_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/max_abs":0.095703125,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/mean":-6.057322025299072e-06,"train/train/tensor_act_model_layers_3_mlp_waleed/std":0.4257835015731739,"train/train/layer_model_layers_86/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_self_attn_q_proj/max_abs":6.34375,"train/train/layer_model_layers_19/grad/norm":0.03914795697646009,"train/train/tensor_act_model_layers_42_self_attn/std":0.12305319068326172,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/norm":4.4375,"train/train/tensor_grad_model_layers_34_mlp_waleed_W_u_weight/max_abs":0.00096893310546875,"train/train/tensor_act_model_layers_67_input_layernorm/max_abs":5.21875,"train/train/tensor_act_model_layers_69_mlp_waleed/mean":-0.0009164810180664062,"train/train/tensor_param_model_layers_63_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_waleed_W_u_weight/max_abs":0.000545501708984375,"train/train/layer_model_layers_78/act/std":0.9515871171101512,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/max_abs":0.2353515625,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn/max_abs":7.53125,"train/train/tensor_act_model_layers_7_mlp_waleed_W_u/max_abs":2.609375,"train/train/tensor_param_model_layers_74_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_13/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_mlp_waleed_W_u_weight/std":9.169224988184596e-05,"train/train/tensor_act_model_layers_64/max_abs":24.5,"train/train/tensor_act_model_layers_90_mlp_down_proj/norm":4383.015450001458,"train/train/layer_model_layers_6/grad/norm":0.1072775038587865,"train/train/tensor_grad_model_layers_32_mlp_waleed_W_g_weight/std":4.3156733868917693e-05,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/max_abs":0.0010986328125,"train/train/tensor_act_model_layers_81_mlp_waleed/max_abs":5.5,"train/train/tensor_act_model_layers_72_self_attn_q_proj/std":1.1796877273660402,"train/train/tensor_act_model_layers_25_self_attn_v_proj/max_abs":1.84375,"train/train/tensor_act_model_layers_86_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/norm":4.6875,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/norm":0.012924437537489111,"train/train/tensor_act_model_layers_79_self_attn_v_proj/norm":3317.4899574812293,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/mean":-2.2380845621228218e-08,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/norm":0.028396963110092627,"train/train/layer_model_layers_84/act/mean":0.0031113624572753906,"train/train/tensor_grad_model_layers_29_mlp_waleed_W_u_weight/max_abs":0.000629425048828125,"train/train/layer_model_layers_61/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/std":0.050048828125,"train/train/tensor_act_model_layers_13_self_attn_o_proj/norm":470.45473175346694,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/norm":3.109375,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/norm":0.0007919425449029639,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/std":0.02880859375,"train/train/tensor_grad_model_layers_1_mlp_waleed_W_g_weight/std":0.0007286121799047315,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/max_abs":0.2431640625,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/mean":-0.000270843505859375,"train/train/tensor_act_model_layers_40_post_attention_layernorm/mean":0.0027265548706054688,"train/train/tensor_act_model_layers_24_mlp_waleed_W_g/std":0.259278728010861,"train/train/tensor_param_model_layers_70_mlp_waleed_W_g_weight/mean":0.0001926422119140625,"train/train/tensor_act_model_layers_76_post_attention_layernorm/norm":5792.60473633134,"train/train/layer__model_layers_59/param/norm":20.757558158490728,"train/train/tensor_act_model_layers_78_self_attn_o_proj/norm":1546.8021338793058,"train/train/tensor_act_model_layers_58_mlp/std":0.08215360426990974,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp_waleed_W_g/max_abs":2.46875,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/mean":0.0002956390380859375,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/norm":3.921875,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_waleed_W_u/mean":-0.002643585205078125,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/max_abs":0.0908203125,"train/train/tensor_act_model_layers_16_self_attn_k_proj/std":1.156250184854931,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/std":6.796148062344219e-05,"train/train/tensor_param_model_layers_79_mlp_waleed_W_u_weight/norm":6.90625,"train/train/tensor_act_model_layers_30_input_layernorm/max_abs":5.59375,"train/train/tensor_act_model_layers_55_input_layernorm/mean":0.0007353425025939941,"train/train/tensor_act_model_layers_27_mlp_waleed_W_u/max_abs":1.765625,"train/train/tensor_act_model_layers_74_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_mlp_down_proj/norm":16964.358780560604,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/std":2.970809307621487e-05,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/max_abs":0.00128173828125,"train/train/tensor_param_model_layers_80_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/max_abs":0.2431640625,"train/train/layer_model_layers_81/grad/std":8.308265010001857e-05,"train/train/layer__model_layers_83/param/std":0.05879853564686469,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/max_abs":0.251953125,"train/train/tensor_act_model_layers_55_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/mean":1.0088086128234863e-05,"train/train/layer_model_layers_66/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/std":0.003408239615214147,"train/train/layer__model_layers_61/param/max_abs":1,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/max_abs":0.189453125,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn_v_proj/norm":2172.5774552020544,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_57_mlp/norm":491.6653339683475,"train/train/tensor_act_model_layers_32_post_attention_layernorm/std":1.0000003631393408,"train/train/tensor_act_model_layers_83/mean":0.0391845703125,"train/train/tensor_act_model_layers_19_post_attention_layernorm/norm":5792.607055666783,"train/train/tensor_act_model_layers_71_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn_q_proj/max_abs":5.46875,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/std":0.041259765625,"train/train/tensor_act_model_layers_12_input_layernorm/norm":5792.611816406939,"train/train/tensor_grad_model_layers_25_mlp_waleed_W_g_weight/std":4.19096587329745e-05,"train/train/tensor_param_model_layers_88_mlp_waleed_W_u_weight/max_abs":0.251953125,"train/train/layer_model_layers_73/act/norm":21084.378466734557,"train/train/tensor_act_model_layers_90_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_v_proj/max_abs":4.15625,"train/train/tensor_param_model_layers_75_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_waleed_W_u_weight/mean":1.2601958587765694e-08,"train/train/tensor_grad_model_layers_48_mlp_waleed_W_g_weight/norm":0.015335287975251013,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/std":0.0245361328125,"train/train/tensor_act_model_layers_48_self_attn_o_proj/mean":-0.00635528564453125,"train/train/layer_model_layers_79/act/std":0.9858474397695065,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/std":2.056512180414057e-05,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/mean":-1.714215613901615e-08,"train/train/tensor_act_model_layers_36_mlp_down_proj/std":0.06243905876614992,"train/train/tensor_act_model_layers_37_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp_waleed/max_abs":4,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/max_abs":0.000461578369140625,"train/train/layer_model_layers_68/act/std":0.8656019058599281,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/mean":0.000316619873046875,"train/train/tensor_param_model_layers_5_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp_waleed_W_u/mean":-0.026153564453125,"train/train/layer__model_layers_68/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_waleed_W_g/std":0.27588041526371976,"train/train/tensor_act_model_layers_7_mlp_waleed_W_g/max_abs":2.875,"train/train/layer_model_layers_37/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_mlp_waleed_W_g_weight/std":0.032958984375,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/mean":-0.000125885009765625,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/max_abs":0.0013427734375,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/max_abs":0.00024318695068359375,"train/train/tensor_act_model_layers_32_mlp_waleed/mean":0.0016498565673828125,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_q_proj/max_abs":5.78125,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/norm":0.014131912476922363,"train/train/tensor_act_model_layers_12_self_attn/mean":-0.001926422119140625,"train/train/tensor_act_model_layers_7_mlp/max_abs":1.2265625,"train/train/tensor_act_model_layers_88_self_attn/mean":-0.0026726722717285156,"train/train/tensor_grad_model_layers_65_mlp_waleed_W_g_weight/max_abs":0.000949859619140625,"train/train/tensor_act_model_layers_13_post_attention_layernorm/mean":-0.00644683837890625,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/std":0.05615234375,"train/train/tensor_act_model_layers_77_mlp_waleed/max_abs":5.375,"train/train/tensor_act_model_layers_36_mlp_down_proj/mean":0.0010528564453125,"train/train/tensor_act_model_layers_9_input_layernorm/norm":5792.611938487495,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/norm":0.0033210395970547585,"train/train/tensor_act_model_layers_4_self_attn_o_proj/max_abs":0.76953125,"train/train/tensor_act_model_layers_60_self_attn_q_proj/norm":5333.823003711885,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_o_proj/norm":530.6829602877826,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_waleed_W_g/norm":3858.17617105795,"train/train/tensor_param_model_layers_3_mlp_waleed_W_u_weight/std":0.0242919921875,"train/train/tensor_act_model_layers_64_self_attn_k_proj/mean":0.015350341796875,"train/train/tensor_act_model_layers_59_mlp_waleed/std":0.13598699436846373,"train/train/tensor_act_model_layers_32_self_attn_o_proj/std":0.024964390894092397,"train/train/tensor_grad_model_layers_90_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43/std":2.6250252460018713,"train/train/tensor_param_model_layers_8_mlp_waleed_W_g_weight/mean":-1.4424324035644531e-05,"train/train/tensor_param_model_layers_50_mlp_waleed_W_u_weight/mean":7.915496826171875e-05,"train/train/tensor_act_model_layers_72_post_attention_layernorm/mean":0.0011148452758789062,"train/train/tensor_param_model_layers_40_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_62_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_49_mlp_waleed_W_g_weight/std":0.0269775390625,"train/train/tensor_param_model_layers_45_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_self_attn_k_proj/mean":-0.026641845703125,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_q_proj/max_abs":7.5625,"train/train/tensor_act_model_layers_27_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_mlp_waleed_W_g_weight/std":5.096487588745287e-05,"train/train/tensor_act_model_layers_14_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn/max_abs":0.57421875,"train/train/tensor_grad_model_layers_89_mlp_waleed_W_g_weight/std":9.336255727496259e-05,"train/train/tensor_act_model_layers_48/mean":0.0026683807373046875,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/mean":1.055002212524414e-05,"train/train/tensor_grad_model_layers_62_mlp_waleed_W_g_weight/mean":9.010545909404755e-08,"train/train/tensor_param_model_layers_72_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_mlp_waleed/std":0.11792018455395037,"train/train/tensor_act_model_layers_51/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_v_proj/max_abs":5.46875,"train/train/tensor_act_model_layers_17_mlp_down_proj/norm":409.8246234998011,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/mean":-4.458427429199219e-05,"train/train/tensor_act_model_layers_0_self_attn_v_proj/norm":4140.77085587096,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/std":3.942415104573992e-05,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/norm":0.008012070337966404,"train/train/tensor_act_model_layers_82_mlp_waleed_W_g/norm":4971.075104084066,"train/train/layer_model_layers_44/grad/mean":-5.7299585080183984e-08,"train/train/tensor_grad_model_layers_67_mlp_waleed_W_g_weight/norm":0.0200295772002796,"train/train/tensor_param_model_layers_32_mlp_waleed_W_g_weight/mean":0.0001678466796875,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/mean":-0.0003566741943359375,"train/train/tensor_act_model_layers_49/std":2.6406499461315924,"train/train/tensor_act_model_layers_23_self_attn_v_proj/mean":0.04150390625,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_5/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_mlp_down_proj/max_abs":5.65625,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_24_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/mean":-0.0003662109375,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/std":3.8281975696382326e-05,"train/train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/std":3.445511551899059e-05,"train/train/tensor_act_model_layers_59_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/std":8.908364073608023e-05,"train/train/tensor_param_model_layers_53_mlp_waleed_W_u_weight/max_abs":0.1376953125,"train/train/layer__model_layers_46/param/std":0.04883107840760211,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/max_abs":0.2451171875,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/norm":0.023730434213621022,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/max_abs":0.000247955322265625,"train/train/tensor_param_model_layers_0_mlp_waleed_W_u_weight/max_abs":0.251953125,"train/train/tensor_act_model_layers_83_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_49/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/mean":-5.206093192100525e-06,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/std":1.9059494379477868e-05,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/norm":0.007735122514494548,"train/train/tensor_act_model_layers_20/std":2.8047420115186763,"train/train/layer_model_layers_69/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_input_layernorm/std":1.0000000169966368,"train/train/tensor_act_model_layers_67_self_attn_q_proj/max_abs":6.25,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/norm":0.006265084867669416,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/norm":7.5625,"train/train/tensor_grad_model_layers_39_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn_k_proj/norm":7366.837939833654,"train/train/tensor_grad_model_layers_5_mlp_waleed_W_u_weight/std":9.682321783075158e-05,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/mean":4.673004150390625e-05,"train/train/tensor_act_model_layers_40_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/max_abs":0.0001277923583984375,"train/train/tensor_param_model_layers_77_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/mean":-5.327165126800537e-06,"train/train/tensor_act_model_layers_29_input_layernorm/mean":-0.00014400482177734375,"train/train/tensor_act_model_layers_79_self_attn/max_abs":3.78125,"train/train/tensor_act_model_layers_29_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_v_proj/mean":-1.3828277587890625e-05,"train/train/layer__model_layers_28/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_mlp_waleed/mean":-0.0010547637939453125,"train/train/tensor_param_model_layers_35_mlp_waleed_W_u_weight/mean":6.246566772460938e-05,"train/train/tensor_act_model_layers_7_self_attn_v_proj/norm":2122.127444046564,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/mean":6.05359673500061e-07,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_k_proj/std":1.2539125053154663,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/mean":0.00020122528076171875,"train/train/layer__model_layers_39/param/std":0.05008076695582009,"train/train/tensor_act_model_layers_61_mlp_waleed/max_abs":3.640625,"train/train/tensor_act_model_layers_43_mlp_waleed_W_g/max_abs":2.6875,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/std":0.0289306640625,"train/train/tensor_act_model_layers_36_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_o_proj/norm":297.90934763965004,"train/train/tensor_act_lm_head/frac_near_dtype_limit":0,"train/train/layer_model_layers_70/grad/mean":-1.1807165214684377e-08,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/norm":0.02280535141755043,"train/train/tensor_act_model_layers_36_self_attn_v_proj/std":0.5195314561513621,"train/train/tensor_act_model_layers_71_mlp_waleed_W_g/mean":-0.00472259521484375,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/std":7.534255622661121e-05,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/norm":0.02013688465719762,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_input_layernorm/norm":5792.616210939568,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/mean":4.369020462036133e-05,"train/train/tensor_grad_model_layers_35_mlp_waleed_W_u_weight/norm":0.013763867559404695,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/norm":0.1187811258581848,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_q_proj/std":0.8164062568445524,"train/train/tensor_act_model_layers_61_self_attn_v_proj/mean":0.00732421875,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/max_abs":0.000762939453125,"train/train/tensor_act_model_layers_75_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/std":0.038330078125,"train/train/tensor_act_model_layers_24/mean":0.0082855224609375,"train/train/tensor_param_model_layers_79_mlp_waleed_W_g_weight/std":0.0380859375,"train/train/tensor_act_model_layers_25_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/max_abs":0.19140625,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/mean":-3.553926944732666e-06,"train/train/tensor_grad_model_layers_61_mlp_waleed_W_g_weight/mean":5.122274160385132e-08,"train/train/tensor_act_model_layers_49_mlp_down_proj/mean":0.00022912025451660156,"train/train/tensor_act_model_layers_57_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_80/act/mean":0.005356788635253906,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp_waleed/max_abs":3.359375,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/std":3.558453578637206e-05,"train/train/tensor_act_model_layers_46_self_attn_o_proj/std":0.0885019313312692,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/max_abs":0.193359375,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_mlp_waleed_W_u_weight/mean":-8.153915405273438e-05,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/max_abs":0.09619140625,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/norm":0.0706899894188309,"train/train/tensor_act_model_layers_60_self_attn/mean":-5.620718002319336e-05,"train/train/tensor_act_model_layers_14_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_input_layernorm/max_abs":4.40625,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/mean":-1.4097895473241806e-07,"train/train/tensor_act_model_layers_65_mlp_waleed_W_g/std":0.4042968773035612,"train/train/tensor_act_model_layers_57_input_layernorm/norm":5792.612182619394,"train/train/tensor_act_model_layers_23_post_attention_layernorm/max_abs":5.21875,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/mean":3.03611159324646e-07,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/max_abs":0.23046875,"train/train/layer__model_layers_5/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_q_proj/max_abs":6.53125,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/max_abs":0.000553131103515625,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/mean":1.6920967027544975e-07,"train/train/tensor_param_model_layers_25_mlp_waleed_W_u_weight/norm":4.34375,"train/train/tensor_act_model_layers_64_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_mlp_waleed_W_g_weight/mean":0.0002593994140625,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/max_abs":0.1416015625,"train/train/tensor_act_model_layers_44_self_attn_v_proj/max_abs":2.71875,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/mean":-0.0004177093505859375,"train/train/tensor_act_model_layers_71_mlp_waleed_W_g/max_abs":3.109375,"train/train/tensor_act_model_layers_29_mlp_waleed/norm":640.9154120868299,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/max_abs":0.00106048583984375,"train/train/tensor_act_model_layers_81_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/max_abs":0.000598907470703125,"train/train/tensor_act_model_layers_39_mlp_waleed_W_g/mean":-0.0081024169921875,"train/train/tensor_act_model_layers_58_self_attn_v_proj/norm":2898.7332706132884,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/norm":0.004624546025523212,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4/frac_near_dtype_limit":0,"train/train/layer_model_layers_75/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_48/act/max_abs":24,"train/train/tensor_act_model_layers_45_mlp/max_abs":0.7109375,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/std":0.058837890625,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_5/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn_k_proj/mean":0.05792236328125,"train/train/tensor_act_model_layers_90_self_attn/std":0.6064731943551857,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/max_abs":0.000919342041015625,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/mean":0.0005340576171875,"train/train/tensor_act_model_layers_59_mlp_down_proj/max_abs":0.6328125,"train/train/tensor_act_model_layers_76_mlp/mean":-0.00092315673828125,"train/train/tensor_act_model_layers_10_post_attention_layernorm/std":1.0000000393483781,"train/train/tensor_act_model_layers_63_self_attn/std":0.17163560907229375,"train/train/tensor_act_model_layers_72_post_attention_layernorm/std":1.0000008411890222,"train/train/tensor_act_model_layers_83_mlp_waleed_W_u/mean":0.031829833984375,"train/train/tensor_param_model_layers_22_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/mean":3.396417014300823e-08,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_post_attention_layernorm/std":1.0000009700010948,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/std":7.18938464700902e-05,"train/train/tensor_act_model_layers_60_mlp/frac_near_user_limit":0,"train/train/layer__model_layers_65/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_70/act/mean":0.013985395431518555,"train/train/layer_model_layers_59/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_v_proj/norm":2622.479215419659,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/mean":2.8848648071289062e-05,"train/train/tensor_act_model_layers_82_post_attention_layernorm/mean":0.017578125,"train/train/tensor_act_model_layers_36_input_layernorm/norm":5792.605957034676,"train/train/tensor_act_model_layers_55_mlp_waleed_W_u/norm":3001.7155641333984,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/mean":0.00020885467529296875,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_self_attn_v_proj/norm":2425.450807699275,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/max_abs":0.22265625,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/max_abs":0.000705718994140625,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60/std":2.593775410484602,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_77_self_attn_q_proj/mean":0.0181884765625,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/std":1.7668130496851614e-05,"train/train/tensor_act_model_layers_47_mlp/max_abs":0.640625,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/std":6.92830864089798e-05,"train/train/tensor_act_model_layers_40_mlp_waleed/norm":898.9677745981238,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/max_abs":0.000522613525390625,"train/train/layer_model_layers_66/grad/mean":2.397937691434869e-08,"train/train/tensor_act_model_layers_31_mlp_down_proj/mean":0.000499725341796875,"train/train/layer__model_layers_84/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_mlp_waleed_W_g/norm":7751.623312167048,"train/train/tensor_act_model_layers_19_mlp_waleed/std":0.09082032309504544,"train/train/tensor_act_model_layers_20_mlp/std":0.05694590028195003,"train/train/tensor_act_model_layers_72_mlp/std":0.14111413048472987,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp_waleed/std":0.16210939026129845,"train/train/tensor_act_model_layers_24_post_attention_layernorm/max_abs":5.375,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/std":2.7000366138881213e-05,"train/train/layer_model_layers_56/grad/max_abs":0.00118255615234375,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/max_abs":0.00118255615234375,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/std":2.9820563399357236e-05,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/max_abs":0.000843048095703125,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/std":6.286952440533862e-05,"train/train/layer_model_layers_18/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68/norm":15581.137479274888,"train/train/layer__model_layers_79/param/mean":0.0014718423209584633,"train/train/tensor_act_model_layers_92_mlp/norm":7902.898941645324,"train/train/tensor_grad_model_layers_60_mlp_waleed_W_u_weight/mean":6.335903890430927e-08,"train/train/tensor_act_model_layers_67_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_waleed/norm":741.5850075583235,"train/train/tensor_act_model_layers_47/mean":0.01216888427734375,"train/train/tensor_act_model_layers_24_post_attention_layernorm/norm":5792.001953126852,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_waleed_W_u_weight/norm":4.15625,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/mean":-8.392333984375e-05,"train/train/tensor_act_model_layers_10_mlp_waleed_W_u/max_abs":2.90625,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/max_abs":0.00086212158203125,"train/train/tensor_grad_model_layers_24_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/max_abs":0.000751495361328125,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_mlp_waleed_W_u_weight/std":0.05810546875,"train/train/tensor_act_model_layers_53_mlp_waleed/std":0.12622141793196667,"train/train/tensor_act_model_layers_47_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/std":6.748324259968119e-05,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/std":4.872005611036236e-05,"train/train/tensor_act_model_layers_31/std":2.687523430318984,"train/train/layer__model_layers_38/param/max_abs":1,"train/train/tensor_act_model_layers_36_self_attn_q_proj/mean":0.0592041015625,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/std":0.04248046875,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/max_abs":0.00023365020751953125,"eval/runtime":17.0629,"train/train/layer_model_layers_75/act/max_abs":25.75,"train/train/tensor_act_model_layers_86_post_attention_layernorm/max_abs":5.65625,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/norm":0.0006984803403846391,"train/train/tensor_act_model_layers_72/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/mean":-0.0003662109375,"train/train/tensor_act_model_layers_81_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/norm":0.01727564490752741,"train/train/tensor_param_model_layers_2_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/mean":-1.7229467630386353e-08,"train/train/layer__model_layers_40/param/norm":19.568944874868063,"train/train/tensor_act_model_layers_25_self_attn_o_proj/std":0.06812238353290181,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_60/grad/norm":0.03148330746585667,"train/train/tensor_act_model_layers_52_self_attn/mean":-0.0006847381591796875,"train/train/tensor_param_model_layers_9_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_mlp_down_proj/mean":0.00528717041015625,"train/train/tensor_act_model_layers_7_mlp_waleed_W_u/mean":-0.0013217926025390625,"train/train/layer_model_layers_12/act/norm":21684.366021700378,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/mean":1.2549571692943573e-07,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_49_self_attn_v_proj/std":0.3090832316268343,"train/train/tensor_act_model_layers_78_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/norm":0.0009670182327489432,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_19_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/mean":1.4994293451309204e-07,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/norm":0.0121379039395666,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/std":0.00014950427713475726,"train/train/tensor_act_model_layers_20/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/max_abs":0.000644683837890625,"train/train/tensor_act_model_layers_68_mlp_waleed_W_g/max_abs":3.171875,"train/train/tensor_param_model_layers_67_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_3_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_37_mlp/std":0.06140153284789958,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_q_proj/std":1.011726375108322,"train/train/layer_model_layers_54/act/norm":19144.65046729852,"train/train/tensor_act_model_layers_86_mlp_waleed/norm":3579.442887283855,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/norm":5.53125,"train/train/tensor_grad_model_layers_3_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/norm":0.012635161880268307,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/norm":4.8125,"train/train/tensor_act_model_layers_47_self_attn/mean":4.034209996461868e-05,"train/train/tensor_param_model_layers_75_mlp_waleed_W_u_weight/norm":6.59375,"train/train/tensor_act_model_layers_8_input_layernorm/norm":5792.6114501964585,"train/train/tensor_grad_model_layers_2_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer_model_layers_36/act/norm":19991.50554064896,"train/train/tensor_act_model_layers_48_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/max_abs":0.0012359619140625,"train/train/layer_model_layers_39/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/norm":0.01150581499465778,"train/train/tensor_act_model_layers_63_mlp_waleed_W_u/norm":3231.9125945440755,"train/train/tensor_param_model_layers_50_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_waleed_W_u/max_abs":12.5625,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/norm":0.028665153172834135,"train/train/tensor_act_model_layers_0_mlp_waleed_W_g/mean":0.0061187744140625,"train/train/layer_model_layers_12/grad/max_abs":0.0028076171875,"train/train/tensor_param_model_layers_15_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/layer_model_layers_70/grad/norm":0.05673481771945416,"train/train/layer_model_layers_80/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/mean":0.00021076202392578125,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/max_abs":0.1650390625,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/mean":-5.459785461425781e-05,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/mean":-3.41096892952919e-08,"train/train/tensor_act_model_layers_85_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_88_mlp_waleed_W_u_weight/mean":2.012820914387703e-07,"train/train/tensor_act_model_norm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/max_abs":0.000461578369140625,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/max_abs":0.0006561279296875,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/mean":-0.0003185272216796875,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/max_abs":0.00110626220703125,"train/train/tensor_param_model_layers_0_mlp_waleed_W_g_weight/mean":-0.00017070770263671875,"train/train/tensor_act_model_layers_87_input_layernorm/norm":5792.609130863712,"train/train/tensor_act_model_layers_81_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_self_attn_o_proj/std":0.19751162106969405,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/norm":0.005539402797936537,"train/train/tensor_param_model_layers_25_mlp_waleed_W_g_weight/max_abs":0.11962890625,"train/train/tensor_act_model_layers_62_mlp_waleed_W_u/max_abs":2.484375,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/norm":0.0038666581383994237,"train/train/tensor_act_model_layers_60_mlp/norm":502.2470797203119,"train/train/tensor_grad_model_layers_93_mlp_waleed_W_u_weight/max_abs":0.00109100341796875,"train/train/tensor_act_model_layers_18_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/std":0.023193359375,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_post_attention_layernorm/norm":5791.096679691895,"train/train/tensor_act_model_layers_15_mlp_waleed/mean":-0.0014591217041015625,"train/train/layer_model_layers_16/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_q_proj/std":1.0312501096138391,"train/train/tensor_act_model_layers_13_self_attn_k_proj/max_abs":5.65625,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/std":4.54920593219078e-05,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/max_abs":0.1474609375,"train/train/layer__model_layers_47/param/max_abs":1,"train/train/tensor_act_model_layers_51_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/mean":-0.0003261566162109375,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/max_abs":0.0009765625,"train/train/tensor_act_model_layers_34_self_attn_q_proj/mean":0.02301025390625,"train/train/tensor_act_model_layers_67_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn_q_proj/norm":5897.064598193235,"train/train/tensor_act_model_layers_61_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp_waleed_W_g/mean":-0.009246826171875,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/std":0.045654296875,"train/train/tensor_param_model_layers_78_mlp_waleed_W_u_weight/std":0.037841796875,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_mlp_waleed_W_g_weight/norm":5.25,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/mean":3.0710361897945404e-07,"train/train/tensor_act_model_layers_40_post_attention_layernorm/max_abs":5.28125,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/max_abs":0.006072998046875,"train/train/tensor_act_model_layers_76_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_q_proj/norm":6844.341998011681,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/max_abs":0.00079345703125,"train/train/tensor_act_model_layers_63_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/mean":0.0001010894775390625,"train/train/tensor_grad_model_layers_39_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_9/grad/std":6.818143730305716e-05,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/mean":-3.87728214263916e-05,"train/train/tensor_act_model_layers_5_mlp_down_proj/std":0.15747127311071102,"train/train/tensor_act_model_layers_68_self_attn_q_proj/max_abs":5.59375,"train/train/tensor_act_model_layers_72_self_attn_v_proj/max_abs":4.5625,"train/train/tensor_param_model_layers_6_mlp_waleed_W_g_weight/mean":6.151199340820312e-05,"train/train/tensor_act_model_layers_92_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/norm":6.34375,"train/train/tensor_grad_model_layers_92_mlp_waleed_W_g_weight/max_abs":0.00128173828125,"train/train/tensor_act_model_layers_66_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_mlp_waleed_W_g_weight/std":4.202245034285509e-05,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_mlp_waleed_W_g_weight/std":5.1204986750139646e-05,"train/train/tensor_act_model_layers_63_mlp/mean":0.000247955322265625,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_65/grad/max_abs":0.00112152099609375,"train/train/tensor_act_model_layers_56_self_attn_o_proj/norm":734.9091898436315,"train/train/tensor_act_model_layers_33_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn/mean":-0.0005855560302734375,"train/train/tensor_act_model_layers_2_self_attn_v_proj/norm":1932.4128191216987,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/norm":0.006517807421196007,"train/train/tensor_act_model_layers_63_self_attn_v_proj/max_abs":3.4375,"train/train/tensor_act_model_layers_55_self_attn_v_proj/max_abs":2.140625,"train/train/tensor_act_model_layers_83_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91/mean":-0.004955291748046875,"train/train/tensor_param_model_layers_68_mlp_waleed_W_g_weight/mean":-0.0001430511474609375,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/std":0.0003573252303261076,"train/train/tensor_act_model_layers_68_self_attn_q_proj/std":0.9960977282733923,"train/train/tensor_param_model_layers_3_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/std":0.0238037109375,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_o_proj/std":0.30127392107565254,"train/train/tensor_act_model_layers_43_self_attn_q_proj/norm":5447.899835649599,"train/train/tensor_act_model_layers_91_mlp_down_proj/max_abs":9.25,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/mean":5.561858415603638e-06,"train/train/tensor_act_model_layers_66_mlp_waleed_W_u/max_abs":2.765625,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_v_proj/norm":1731.3540097039156,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/norm":0.01715837091006734,"train/train/tensor_param_model_layers_55_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_15_post_attention_layernorm/mean":-0.00589752197265625,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn_v_proj/norm":1935.9501414462895,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/std":0.02734375,"train/train/tensor_act_model_layers_31_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_waleed_W_u_weight/mean":2.789311110973358e-07,"train/train/layer_model_layers_4/act/std":1.0705499555379194,"train/train/layer_model_layers_53/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_mlp_waleed_W_g_weight/norm":4.15625,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/std":3.5880511538647496e-05,"train/train/tensor_act_model_layers_24_self_attn_v_proj/std":0.40332164506931667,"train/train/tensor_grad_model_layers_47_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/std":9.813595044633978e-05,"train/train/layer__model_layers_2/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_o_proj/norm":379.34823661869825,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/norm":0.015020415896392471,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/std":4.804270516518702e-05,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/norm":7.03125,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/norm":5.625,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/norm":0.029360765092311163,"train/train/tensor_act_model_layers_83_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_64/param/std":0.052467080269961996,"train/train/tensor_act_model_layers_69_mlp_waleed/frac_near_user_limit":0,"train/train/layer__model_layers_81/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/mean":-3.247987478971481e-08,"train/train/tensor_grad_model_layers_33_mlp_waleed_W_g_weight/std":3.18310019486969e-05,"train/train/tensor_act_model_layers_86_self_attn_q_proj/mean":0.1102294921875,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/mean":-9.862706065177917e-07,"train/train/tensor_act_model_layers_49_self_attn_v_proj/mean":-0.0043487548828125,"train/train/tensor_act_model_layers_18/max_abs":26.125,"train/train/layer_model_layers_86/grad/max_abs":0.0015106201171875,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/std":5.452376012602669e-05,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_waleed/std":0.10620146408722539,"train/train/tensor_act_model_layers_45_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/max_abs":0.00020599365234375,"train/train/tensor_act_model_layers_22_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_v_proj/norm":2039.327014757359,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/mean":2.4330802261829376e-08,"train/train/tensor_act_model_layers_58_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_mlp_waleed_W_u_weight/norm":0.013534709323038333,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/mean":-2.187490463256836e-05,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/std":6.798487056133052e-05,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/mean":0.000110626220703125,"train/train/tensor_act_model_layers_93_mlp_down_proj/mean":-0.03179931640625,"train/train/tensor_act_model_layers_20_post_attention_layernorm/max_abs":5.1875,"train/train/tensor_param_model_layers_51_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/max_abs":0.00138092041015625,"train/train/layer__model_layers_9/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_waleed_W_u_weight/std":0.028076171875,"train/train/tensor_act_model_layers_49_mlp_waleed_W_u/mean":-0.0004055500030517578,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_10_mlp/norm":647.1625773204828,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_mlp_waleed_W_u_weight/std":3.874907204280261e-05,"train/train/tensor_param_model_layers_26_mlp_waleed_W_g_weight/norm":4.3125,"train/train/tensor_act_model_layers_21_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_waleed_W_u/std":0.9062501479839337,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn_k_proj/max_abs":4.15625,"train/train/tensor_param_model_layers_60_mlp_waleed_W_g_weight/mean":-0.00012969970703125,"train/train/tensor_act_model_layers_54_self_attn_o_proj/norm":585.6820045138335,"train/train/tensor_act_model_layers_43_post_attention_layernorm/max_abs":5.21875,"train/train/tensor_param_model_layers_44_mlp_waleed_W_g_weight/mean":6.943941116333008e-06,"train/train/tensor_act_model_layers_92_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp_down_proj/mean":-0.00675201416015625,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/mean":-2.7194619178771973e-07,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/std":2.7204010646980004e-05,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/std":0.04296875,"train/train/tensor_param_model_layers_29_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_mlp_waleed_W_g_weight/std":0.0291748046875,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59/mean":0.0117645263671875,"train/train/layer__model_layers_27/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/max_abs":0.1201171875,"train/train/layer_model_layers_61/act/max_abs":24.5,"train/train/tensor_act_model_layers_89_post_attention_layernorm/mean":-0.0023784637451171875,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/std":0.0223388671875,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_42/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp_waleed/max_abs":2.84375,"train/train/tensor_param_model_layers_55_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/max_abs":0.1318359375,"train/train/tensor_grad_model_layers_86_mlp_waleed_W_g_weight/std":8.979405644268667e-05,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/mean":0.0002288818359375,"train/train/tensor_act_model_layers_57_self_attn/mean":0.0006399154663085938,"train/train/tensor_grad_model_layers_8_mlp_waleed_W_g_weight/norm":0.025339074255013833,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/std":4.430899349541969e-05,"train/train/tensor_act_model_layers_39_self_attn_o_proj/max_abs":1.6171875,"train/train/layer__model_layers_77/param/norm":23.068681420922175,"train/train/tensor_act_model_layers_53_mlp_waleed_W_g/norm":2925.154358256927,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/norm":0.0011014306681962599,"train/train/layer_model_layers_76/act/frac_near_user_limit":0,"train/train/layer_model_layers_7/grad/mean":-3.633066116833649e-08,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_37/param/norm":19.846739440176314,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/std":5.210605997295656e-05,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/max_abs":0.271484375,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/std":0.023681640625,"train/train/tensor_act_model_layers_91_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp/max_abs":0.66796875,"train/train/tensor_grad_model_layers_16_mlp_waleed_W_g_weight/max_abs":0.00066375732421875,"train/train/tensor_act_model_layers_4_mlp_down_proj/max_abs":1.3984375,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73/mean":0.0008525848388671875,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_87/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_mlp_waleed_W_g_weight/max_abs":0.1455078125,"train/train/tensor_param_model_layers_22_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_v_proj/norm":4194.707377727822,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/max_abs":0.1552734375,"train/train/tensor_act_model_layers_41_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_93/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/norm":6.75,"train/train/tensor_act_model_layers_8/std":3.0937699506620144,"train/train/layer_model_layers_28/act/std":0.8873481856435738,"train/train/tensor_act_model_layers_80/max_abs":26.875,"train/train/tensor_grad_model_layers_42_mlp_waleed_W_g_weight/mean":-2.5305780582129955e-08,"train/train/tensor_param_model_layers_13_mlp_waleed_W_u_weight/mean":-1.6167759895324707e-06,"train/train/tensor_param_model_layers_49_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/std":3.7807992835931184e-05,"train/train/tensor_grad_model_layers_93_mlp_waleed_W_u_weight/std":0.00012566536924650933,"train/train/tensor_act_model_layers_43_mlp_waleed_W_g/std":0.31445325263165913,"train/train/tensor_act_model_layers_51_post_attention_layernorm/norm":5792.610229493622,"train/train/tensor_act_model_layers_32_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/max_abs":0.000583648681640625,"train/train/tensor_param_model_layers_28_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_input_layernorm/max_abs":5.40625,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/norm":0.017458828792354622,"train/train/tensor_grad_model_layers_23_mlp_waleed_W_g_weight/std":3.523012759768348e-05,"train/train/tensor_grad_model_layers_9_mlp_waleed_W_u_weight/max_abs":0.00138092041015625,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_42_self_attn/mean":0.0008697509765625,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/norm":0.0005307203997173685,"train/train/tensor_act_model_layers_53_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp/norm":2825.899297889857,"train/train/tensor_param_model_layers_58_mlp_waleed_W_u_weight/std":0.0291748046875,"train/train/tensor_grad_model_layers_58_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp_waleed_W_g/mean":-0.0014171600341796875,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/mean":9.489059448242188e-05,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/mean":9.918212890625e-05,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/mean":2.866145223379135e-07,"train/train/tensor_act_model_layers_5_input_layernorm/std":1.0000000964791933,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/std":0.04248046875,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/mean":3.647059202194214e-06,"train/train/tensor_act_model_layers_70_self_attn_k_proj/max_abs":5.375,"train/train/layer__model_layers_9/param/std":0.04754293007792882,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/mean":0.00022029876708984375,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/max_abs":0.0008544921875,"train/train/layer__model_layers_10/param/max_abs":1,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_embed_tokens/max_abs":0.5,"train/train/tensor_act_model_layers_55_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_mlp_waleed_W_u_weight/std":0.0247802734375,"train/train/tensor_grad_model_layers_68_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/std":5.84349378426918e-05,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/std":0.00018017085875978974,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/mean":-4.267692565917969e-05,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/mean":0.00014781951904296875,"train/train/tensor_act_model_layers_84_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/norm":0.026434494578200074,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/mean":-0.000682830810546875,"train/train/layer_model_layers_34/act/max_abs":26,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_waleed/max_abs":5.28125,"train/train/tensor_act_model_layers_73_mlp/std":0.15405332950816295,"train/train/tensor_param_model_layers_33_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/mean":4.452886059880257e-06,"train/train/tensor_act_model_layers_35_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/max_abs":0.185546875,"train/train/tensor_param_model_layers_22_mlp_waleed_W_g_weight/max_abs":0.11328125,"train/train/tensor_param_model_layers_28_mlp_waleed_W_u_weight/std":0.0242919921875,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/std":0.04150390625,"train/train/tensor_act_model_layers_87/norm":20745.799910969934,"train/train/tensor_act_model_layers_77_mlp_down_proj/std":0.18920950420771576,"train/train/tensor_act_model_layers_32_self_attn_k_proj/norm":4695.061652023998,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/max_abs":0.13671875,"train/train/tensor_act_model_layers_76_self_attn_o_proj/norm":1431.7324613888693,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp_waleed/std":0.07800321700071386,"train/train/tensor_param_model_layers_60_mlp_waleed_W_g_weight/max_abs":0.142578125,"train/train/tensor_grad_model_layers_10_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_mlp_waleed_W_u/mean":0.00373077392578125,"train/train/tensor_act_model_layers_64_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_mlp_waleed_W_g/norm":3452.3949279420317,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/max_abs":0.1357421875,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/mean":2.5704503059387207e-07,"train/train/tensor_act_model_layers_68_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp/max_abs":0.69140625,"train/train/tensor_act_model_layers_64_self_attn_q_proj/mean":0.069580078125,"train/train/tensor_act_model_layers_30_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_32_mlp_waleed_W_u_weight/max_abs":0.000598907470703125,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp/max_abs":24.5,"train/train/tensor_act_model_layers_29_self_attn_q_proj/mean":-0.01849365234375,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/mean":0.0003223419189453125,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/std":7.544062877629638e-05,"train/train/tensor_act_model_layers_53/mean":0.007251739501953125,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/mean":-0.0003719329833984375,"train/train/tensor_act_model_layers_60_self_attn_o_proj/max_abs":1.2421875,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/mean":6.437301635742188e-05,"train/train/tensor_act_model_layers_13_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_72/grad/norm":0.06292704513732317,"train/train/tensor_act_model_layers_31_mlp_waleed_W_u/norm":2301.4179016855137,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/std":0.026123046875,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/norm":0.005099107418829161,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/mean":-2.5596818886697292e-08,"train/train/tensor_act_model_layers_3/frac_near_dtype_limit":0,"train/train/layer__model_layers_51/param/max_abs":1,"train/train/tensor_act_model_layers_7_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp_waleed_W_g/max_abs":5.15625,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/mean":0.0003070831298828125,"train/train/tensor_param_model_layers_88_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_8_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/std":4.5062622458689765e-05,"train/train/layer_model_layers_45/act/norm":19688.941873419924,"train/train/tensor_act_model_layers_84_self_attn_q_proj/max_abs":6.8125,"train/train/tensor_act_model_layers_90_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_79/grad/mean":1.6181693410910607e-07,"train/train/tensor_act_model_layers_28_post_attention_layernorm/mean":-0.0001850128173828125,"train/train/tensor_act_model_layers_68_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_v_proj/norm":2925.3252786110274,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/std":8.894835980747044e-05,"train/train/layer__model_layers_63/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_14/grad/max_abs":0.00616455078125,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/max_abs":0.1259765625,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/norm":0.006725578497534556,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/mean":4.839897155761719e-05,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/norm":0.0021378919407321053,"train/train/tensor_act_model_layers_66_mlp_waleed_W_u/std":0.4013685124322041,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/std":0.029541015625,"train/train/layer_model_layers_30/grad/std":5.364153786052769e-05,"train/train/tensor_grad_model_embed_tokens_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn/norm":994.2581249560949,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/mean":9.047985076904297e-05,"train/train/tensor_act_model_layers_32_self_attn/mean":-0.0002598762512207031,"train/train/tensor_act_model_layers_51_post_attention_layernorm/std":1.0000013646070796,"train/train/tensor_grad_model_layers_0_mlp_waleed_W_g_weight/max_abs":0.0380859375,"train/train/tensor_act_model_layers_73/max_abs":25.75,"train/train/tensor_act_model_layers_1_mlp_waleed/max_abs":13.375,"train/train/layer__model_layers_18/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/norm":3.75,"train/train/tensor_param_model_layers_3_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/max_abs":0.00110626220703125,"train/train/tensor_param_model_layers_1_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/norm":3.53125,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/max_abs":0.0023193359375,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/norm":0.006888544790438464,"train/train/tensor_act_model_layers_75_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_waleed_W_g_weight/norm":5.1875,"train/train/tensor_param_model_layers_42_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/mean":-0.0004863739013671875,"train/train/layer__model_layers_21/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn_q_proj/norm":5759.700997711279,"train/train/layer__model_layers_20/param/norm":19.429466034624447,"train/train/tensor_act_model_layers_69_self_attn/mean":-0.0004891157150268555,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/max_abs":0.000408172607421875,"train/train/tensor_act_model_layers_11_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_input_layernorm/norm":5792.616333011366,"train/train/tensor_param_model_layers_10_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/std":0.046875,"train/train/tensor_act_model_layers_62_mlp_down_proj/mean":0.00017702579498291016,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/max_abs":0.00013637542724609375,"train/train/layer_model_layers_88/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp_waleed_W_u/max_abs":3.515625,"train/train/tensor_act_model_layers_63_self_attn_o_proj/mean":0.0005640983581542969,"train/train/tensor_act_model_layers_25_mlp_waleed_W_g/norm":2202.9910689128183,"train/train/tensor_act_model_layers_39_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_post_attention_layernorm/max_abs":4.5625,"train/train/tensor_act_model_layers_69_self_attn_q_proj/norm":6048.11104524117,"train/train/tensor_act_model_layers_58_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_waleed/norm":743.9215561186331,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/std":0.0235595703125,"train/train/layer__model_layers_55/param/max_abs":1,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn_o_proj/max_abs":0.953125,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/max_abs":0.0013275146484375,"train/train/layer_model_layers_35/grad/norm":0.03063333658161598,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_63/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_input_layernorm/norm":5792.607666018429,"train/train/tensor_act_model_layers_35_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_waleed_W_u_weight/std":0.034423828125,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_77_self_attn_k_proj/max_abs":6.625,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/mean":-1.1261727195233107e-07,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_36/act/max_abs":25.375,"train/train/tensor_act_model_layers_30_mlp_waleed_W_u/norm":2257.557491889345,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/max_abs":0.00013065338134765625,"train/train/tensor_act_model_layers_50_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80/std":3.1367266234445155,"train/train/tensor_grad_model_layers_30_mlp_waleed_W_g_weight/std":3.660405616316297e-05,"train/train/tensor_grad_model_layers_46_mlp_waleed_W_u_weight/norm":0.014191362917996945,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/norm":4.46875,"train/train/tensor_act_model_layers_77_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/max_abs":0.000598907470703125,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/max_abs":0.0002384185791015625,"train/train/tensor_act_model_layers_37_self_attn_v_proj/std":0.3735361511044932,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/norm":2.78125,"train/train/tensor_act_model_layers_43_self_attn/max_abs":1.828125,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/norm":3.3125,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/std":1.6136603049926184e-05,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/mean":-0.0001773834228515625,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/max_abs":0.2314453125,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_v_proj/norm":2419.3637917049673,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/std":0.00017888884826895707,"train/train/tensor_act_model_layers_17_self_attn/mean":-0.00347137451171875,"train/train/tensor_act_model_layers_50_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/std":5.430652219582468e-05,"train/train/tensor_param_model_layers_11_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/max_abs":0.0009918212890625,"train/train/layer_model_layers_16/grad/max_abs":0.0023956298828125,"train/train/tensor_grad_model_layers_41_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_mlp_waleed_W_u_weight/std":0.041748046875,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/std":0.023681640625,"train/train/tensor_act_model_layers_52_mlp/mean":0.0006361007690429688,"train/train/tensor_act_model_layers_43_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_mlp_waleed_W_u_weight/std":4.471763663586829e-05,"train/train/tensor_act_model_layers_80_input_layernorm/std":1.0000004791653498,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/max_abs":0.000789642333984375,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/mean":2.2619962692260742e-05,"train/train/tensor_param_model_layers_61_mlp_waleed_W_u_weight/std":0.02978515625,"train/train/tensor_grad_model_layers_22_mlp_waleed_W_u_weight/max_abs":0.00165557861328125,"train/train/tensor_act_model_layers_26_self_attn_q_proj/std":0.9931657035311454,"train/train/tensor_grad_model_layers_75_mlp_waleed_W_u_weight/mean":-1.1503198038553819e-07,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/max_abs":9.250640869140625e-05,"train/train/tensor_act_model_layers_45_mlp_down_proj/norm":389.85733023765897,"train/train/tensor_act_model_layers_17_mlp_waleed_W_g/mean":0.00093841552734375,"train/train/tensor_grad_model_layers_56_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_80/param/norm":23.544505728513393,"train/train/tensor_act_model_layers_35_mlp/norm":349.48217769225516,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_down_proj/max_abs":2.453125,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_k_proj/mean":-0.0064697265625,"train/train/tensor_act_model_layers_59_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_post_attention_layernorm/norm":5792.60559082397,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/max_abs":0.00127410888671875,"train/train/tensor_param_model_layers_44_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/mean":2.1333107724785805e-08,"train/train/tensor_act_model_layers_77_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_31/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/mean":0.0002117156982421875,"train/train/layer_model_layers_84/grad/norm":0.09356113964829363,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_46_mlp_waleed_W_g/std":0.32226565498294113,"train/train/tensor_param_model_layers_53_mlp_waleed_W_u_weight/mean":-2.753734588623047e-05,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_mlp_waleed_W_g_weight/max_abs":0.0006256103515625,"train/train/tensor_grad_model_layers_23_mlp_waleed_W_u_weight/mean":-7.223570719361305e-08,"train/train/tensor_act_model_layers_74_post_attention_layernorm/norm":5792.610839847712,"train/train/tensor_act_model_layers_59_mlp_waleed_W_g/max_abs":2.234375,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/mean":-9.298324584960938e-06,"train/train/tensor_act_model_layers_73/norm":16140.301952745129,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/std":9.7879247965217e-05,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/norm":0.005182555102516817,"train/train/tensor_act_model_layers_1_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/max_abs":0.00014495849609375,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/max_abs":0.000339508056640625,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_self_attn_o_proj/std":0.13378909220374843,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/max_abs":0.00024127960205078125,"train/train/tensor_param_model_layers_68_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn_k_proj/max_abs":4.9375,"train/train/tensor_grad_model_layers_7_mlp_waleed_W_g_weight/std":8.536652290678696e-05,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/norm":4.34375,"train/train/tensor_act_model_layers_43_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_v_proj/std":0.4179688437937471,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/norm":4.28125,"train/train/layer__model_layers_71/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/max_abs":0.23828125,"train/train/tensor_act_model_layers_50_mlp_down_proj/norm":541.6640901350356,"train/train/tensor_act_model_layers_86_mlp/std":0.44775907765431594,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/mean":-9.918212890625e-05,"train/train/tensor_act_model_layers_88_mlp_waleed_W_g/norm":5764.6323261533,"train/train/tensor_act_model_layers_66_self_attn_q_proj/std":1.017583828725224,"train/train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_waleed/mean":0.003810882568359375,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/mean":4.315376281738281e-05,"train/train/tensor_param_model_layers_91_mlp_waleed_W_u_weight/norm":10.5,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/norm":0.030765472667466143,"train/train/tensor_act_model_layers_49_mlp_waleed/std":0.1291510909354246,"train/train/tensor_act_model_layers_65_self_attn_o_proj/norm":738.3666472597946,"train/train/tensor_act_model_layers_67_self_attn/norm":1836.6584070356207,"train/train/tensor_act_model_layers_17_mlp/mean":0.001007080078125,"train/train/layer_model_layers_18/act/std":0.8854606896953838,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/std":0.0004863739013671875,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/norm":0.0009079480357739584,"train/train/tensor_act_model_layers_88_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/std":4.8167418129597896e-05,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/max_abs":0.267578125,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/std":0.00011124909558806248,"train/train/tensor_grad_model_layers_5_mlp_waleed_W_g_weight/max_abs":0.00139617919921875,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_mlp_waleed_W_u_weight/norm":0.017029355494970037,"train/train/tensor_grad_model_layers_47_mlp_waleed_W_g_weight/norm":0.012940532711264661,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/mean":-0.00014019012451171875,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/std":8.884365298165714e-05,"train/train/layer__model_layers_6/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_o_proj/mean":-0.0006694793701171875,"train/train/tensor_act_model_layers_19_mlp_waleed_W_u/mean":-0.005889892578125,"train/train/tensor_grad_model_layers_27_mlp_waleed_W_g_weight/mean":7.33998604118824e-08,"train/train/layer_model_layers_36/grad/std":5.574270282269364e-05,"train/train/tensor_act_model_layers_78_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/std":2.9465206635280213e-05,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/norm":0.01977783941483745,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/mean":-1.7727870726957917e-08,"train/train/tensor_act_model_layers_43_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/max_abs":0.0002803802490234375,"train/train/tensor_act_model_layers_42_mlp/max_abs":0.73046875,"train/train/tensor_param_model_layers_50_mlp_waleed_W_g_weight/std":0.0272216796875,"train/train/layer_model_layers_44/act/std":0.8293111301987549,"train/train/tensor_act_model_layers_13_self_attn_q_proj/norm":7092.830406585878,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/max_abs":0.00109100341796875,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/std":1.9856946040756148e-05,"train/train/tensor_act_model_layers_14/mean":-0.0074615478515625,"train/train/tensor_act_model_layers_40_self_attn/std":0.051453281964750505,"train/train/tensor_act_model_layers_5_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_mlp_waleed_W_g_weight/norm":0.024203267840166492,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_52/grad/max_abs":0.000911712646484375,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/std":8.728730598290495e-05,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/norm":2.890625,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp_waleed/max_abs":4.5,"train/train/tensor_grad_model_embed_tokens_weight/norm":0.5910809407067672,"train/train/tensor_act_model_layers_48_mlp_waleed_W_u/std":0.3520518230225322,"train/train/tensor_act_model_layers_64_mlp_waleed_W_u/max_abs":3.078125,"train/train/tensor_param_model_layers_47_mlp_waleed_W_u_weight/std":0.0264892578125,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/norm":0.000742754962407004,"train/train/tensor_param_model_layers_31_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn_q_proj/norm":6417.63339686645,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/std":5.968006765052245e-05,"train/train/tensor_act_model_layers_0_self_attn_k_proj/max_abs":5.625,"train/train/layer_model_layers_0/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/max_abs":0.0004444122314453125,"train/train/tensor_act_model_layers_9_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/norm":6.84375,"train/train/tensor_act_model_layers_14_self_attn_q_proj/max_abs":4.90625,"train/train/tensor_act_model_layers_79_input_layernorm/norm":5792.617553717136,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/max_abs":0.000537872314453125,"train/train/tensor_act_model_layers_36_self_attn_k_proj/mean":2.7000904083251953e-05,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/std":5.4768732350349406e-05,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/max_abs":0.00098419189453125,"train/train/tensor_act_model_layers_26_self_attn/mean":-0.000213623046875,"train/train/tensor_act_model_layers_51_post_attention_layernorm/max_abs":5.0625,"train/train/tensor_act_model_layers_26_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_37/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_mlp_down_proj/mean":0.0013427734375,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/norm":6.6875,"train/train/tensor_act_model_layers_46_mlp_waleed/norm":953.9761471472153,"train/train/tensor_act_model_layers_18/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_93/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_k_proj/max_abs":5.03125,"train/train/tensor_param_model_layers_26_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/std":5.0064125929082964e-05,"train/train/tensor_act_model_layers_7_self_attn_v_proj/max_abs":1.640625,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/norm":0.013916484088708929,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/norm":4.625,"train/train/tensor_act_model_layers_69_self_attn_q_proj/std":1.0390634966967398,"train/train/layer_model_layers_54/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_93/act/norm":51212.346234818986,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_input_layernorm/max_abs":5.15625,"train/train/tensor_act_model_layers_47_self_attn/std":0.027649560878366185,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/mean":-2.5175977498292923e-06,"train/train/tensor_act_model_layers_40_input_layernorm/std":1.0000011801937347,"train/train/layer_model_layers_70/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/norm":0.017395340765948345,"train/train/tensor_param_model_layers_92_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_27/grad/max_abs":0.00189208984375,"train/train/tensor_act_model_layers_52_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_input_layernorm/std":1.0000000533182158,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/std":2.9265848739418592e-05,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/std":4.7866556485739744e-05,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/std":4.741383524658883e-05,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/max_abs":0.00124359130859375,"train/train/tensor_grad_model_layers_0_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_48_mlp_down_proj/std":0.10132101497846731,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_k_proj/max_abs":5.4375,"train/train/tensor_param_model_layers_15_mlp_waleed_W_g_weight/std":0.0230712890625,"train/train/tensor_act_model_layers_57_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_mlp_waleed/std":0.10266136244250366,"train/train/tensor_act_model_layers_84_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/norm":0.0047940774395456714,"train/train/tensor_param_model_layers_21_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/mean":0.00011014938354492188,"train/train/tensor_param_model_layers_22_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_32_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_post_attention_layernorm/norm":5792.61230469248,"train/train/tensor_act_model_layers_4_mlp_waleed_W_g/mean":-0.009674072265625,"train/train/tensor_act_model_layers_68_self_attn_o_proj/norm":813.8348934941843,"train/train/tensor_act_model_layers_1_self_attn_k_proj/norm":5721.718111799674,"train/train/tensor_grad_model_layers_5_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_down_proj/max_abs":0.66015625,"train/train/layer_model_layers_21/act/max_abs":26.375,"train/train/layer__model_layers_58/param/norm":21.295654082758293,"train/train/tensor_act_model_layers_78_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/norm":0.06330460953192751,"train/train/tensor_act_model_layers_64_mlp_waleed/max_abs":3.609375,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/std":7.230210770551524e-05,"train/train/tensor_act_model_layers_75_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_mlp/std":0.09350681639037162,"train/train/tensor_act_model_layers_77_self_attn_k_proj/mean":-0.0816650390625,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_44_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_waleed_W_g/std":0.39453149668053195,"train/train/tensor_grad_model_layers_49_mlp_waleed_W_g_weight/std":4.107535475548526e-05,"train/train/tensor_act_model_layers_83_input_layernorm/mean":0.0146026611328125,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/max_abs":0.00140380859375,"train/train/tensor_act_model_layers_49_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/mean":-0.00010919570922851562,"train/train/tensor_act_model_layers_58_mlp_waleed/max_abs":3.15625,"train/train/tensor_act_model_layers_28_self_attn_k_proj/norm":6186.069983292181,"train/train/tensor_act_model_layers_43_mlp_waleed/mean":0.0008134841918945312,"train/train/tensor_param_model_layers_14_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/norm":0.007698433359383858,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/std":2.79564878678236e-05,"train/train/layer_model_layers_74/act/norm":22003.706439877064,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_82/grad/norm":0.07703931173746163,"train/train/layer__model_layers_42/param/max_abs":1,"train/train/tensor_act_model_layers_28_self_attn_k_proj/mean":-0.00042247772216796875,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_30_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_86/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/std":9.32745075715948e-05,"train/train/tensor_act_model_layers_52_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_post_attention_layernorm/max_abs":5,"train/train/tensor_act_model_layers_84_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28/std":2.703148175477337,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/norm":5,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_mlp_waleed_W_g_weight/mean":0.00016307830810546875,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/norm":7.09375,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/std":0.0228271484375,"train/train/layer__model_layers_6/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_mlp/norm":1769.409631625082,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/max_abs":0.001617431640625,"train/train/tensor_grad_model_layers_74_mlp_waleed_W_u_weight/max_abs":0.00084686279296875,"train/train/tensor_act_model_layers_91_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/max_abs":0.11181640625,"train/train/layer_model_layers_52/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_waleed/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_post_attention_layernorm/std":1.000000347848921,"train/train/tensor_act_model_layers_20_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn/norm":976.183720747352,"train/train/tensor_grad_model_layers_60_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/std":0.00033070012716623356,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_89/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/std":0.033935546875,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/max_abs":0.08642578125,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/mean":8.32369551062584e-09,"train/train/tensor_act_model_layers_63_self_attn_v_proj/mean":0.00609588623046875,"_step":106,"train/train/tensor_act_model_layers_89_mlp_waleed/max_abs":10.375,"train/train/layer_model_layers_9/grad/max_abs":0.0024261474609375,"train/train/tensor_grad_model_layers_56_mlp_waleed_W_u_weight/norm":0.015111471470555634,"train/train/tensor_act_model_layers_6_input_layernorm/mean":-0.0045166015625,"train/train/tensor_act_model_layers_80_mlp_waleed/norm":2476.4383318086348,"train/train/layer_model_layers_60/act/frac_near_dtype_limit":0,"train/train/layer__model_layers_22/param/std":0.04818908902499146,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/std":5.466635941885701e-05,"train/train/tensor_act_model_layers_35_mlp/mean":0.002880096435546875,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/norm":3.4375,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/max_abs":0.00017452239990234375,"train/train/layer__model_layers_90/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_mlp/mean":0.00653076171875,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn_v_proj/std":0.5000000253785395,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/mean":0.00014591217041015625,"train/train/tensor_act_model_layers_36_self_attn_q_proj/max_abs":6,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_mlp_waleed_W_g_weight/std":0.035400390625,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/norm":0.015763060265687176,"train/train/tensor_act_model_layers_74_mlp_waleed_W_u/mean":-0.0057525634765625,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/mean":-0.0002193450927734375,"train/train/layer__model_layers_7/param/std":0.04750612806150061,"train/train/tensor_param_model_layers_37_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp_waleed_W_g/norm":5808.658460839306,"train/train/layer_model_layers_74/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/norm":0.003775047234332867,"train/train/tensor_param_model_layers_87_mlp_waleed_W_g_weight/max_abs":0.255859375,"train/train/layer_model_layers_61/act/norm":19957.555210077884,"train/train/tensor_act_model_layers_39_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_input_layernorm/mean":0.004055023193359375,"train/train/tensor_param_model_layers_16_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/norm":0.006759867438404081,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/norm":3.078125,"train/train/tensor_act_model_layers_10_self_attn_k_proj/std":1.6171875136054081,"train/train/layer_model_layers_77/act/norm":22450.843634848774,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_mlp_waleed_W_g_weight/mean":1.2367963790893555e-06,"train/train/tensor_act_model_layers_39_mlp/std":0.06335485280031684,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/mean":-0.00012683868408203125,"train/train/tensor_act_model_layers_54_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/norm":5.8125,"train/train/layer__model_layers_20/param/mean":0.0015290449264454954,"train/train/tensor_act_model_layers_22_mlp/max_abs":0.6640625,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/mean":-7.543712854385376e-08,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/std":0.024169921875,"train/train/tensor_act_model_layers_24_self_attn_o_proj/std":0.21997220454830863,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/max_abs":0.000797271728515625,"train/train/tensor_param_model_layers_57_mlp_waleed_W_g_weight/max_abs":0.1357421875,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/norm":0.0009406632741607816,"train/train/tensor_param_model_layers_56_input_layernorm_weight/std":0,"train/train/layer_model_layers_0/grad/mean":-2.6332048841646047e-06,"train/train/tensor_act_model_layers_86_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_waleed_W_u/max_abs":2.421875,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/norm":2.96875,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/mean":-5.5789947509765625e-05,"train/train/layer_model_layers_64/grad/max_abs":0.0015411376953125,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_mlp_waleed_W_u_weight/max_abs":0.00079345703125,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/mean":1.210719347000122e-08,"train/train/tensor_act_model_layers_79_self_attn_v_proj/std":0.5722692598546435,"train/train/tensor_param_model_layers_86_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn/max_abs":4.9375,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_waleed_W_u/mean":0.0177001953125,"train/train/tensor_act_model_layers_35_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_37_mlp_waleed/max_abs":3.109375,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/std":2.381601581311464e-05,"train/train/tensor_act_model_layers_17_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_q_proj/max_abs":7.5625,"train/train/tensor_param_model_layers_36_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/layer__model_layers_88/param/max_abs":1,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/max_abs":0.0023956298828125,"train/train/tensor_grad_model_layers_60_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/mean":8.195638656616211e-06,"train/train/tensor_act_model_layers_60_self_attn_k_proj/std":0.8330096319420232,"train/train/tensor_param_model_layers_51_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_27_input_layernorm_weight/mean":1,"train/train/layer__model_layers_72/param/norm":22.200947729277797,"train/train/tensor_grad_model_layers_87_mlp_waleed_W_u_weight/mean":-2.816959749907255e-07,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/mean":2.4080276489257812e-05,"train/train/tensor_act_model_layers_39_self_attn/std":0.14868266414880968,"train/train/tensor_act_model_layers_85_self_attn_q_proj/mean":0.207275390625,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_31_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_input_layernorm/mean":-0.0020704269409179688,"train/train/tensor_act_model_layers_60_self_attn_o_proj/norm":406.7039494663982,"train/train/tensor_act_model_layers_15_self_attn_q_proj/norm":6462.419515749725,"train/train/tensor_act_model_layers_82_mlp_waleed_W_u/mean":-0.0047607421875,"train/train/tensor_act_model_layers_50_mlp_waleed_W_g/max_abs":2.875,"train/train/tensor_act_model_layers_48_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_v_proj/max_abs":4.375,"train/train/tensor_act_model_layers_5_mlp/mean":0.0078582763671875,"train/train/tensor_grad_model_layers_70_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_mlp/mean":0.00018596649169921875,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/max_abs":0.1669921875,"train/train/layer__model_layers_71/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_1/param/max_abs":1,"train/train/layer_model_layers_22/grad/max_abs":0.00165557861328125,"train/train/tensor_act_model_layers_81_post_attention_layernorm/std":1.0000002277083435,"train/train/tensor_act_model_layers_67_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp_waleed/norm":671.9986306641081,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/layer_model_layers_41/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/max_abs":0.2451171875,"train/train/tensor_act_model_layers_90_post_attention_layernorm/max_abs":5.75,"train/train/tensor_act_model_layers_5_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/norm":3.1875,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/std":0.046142578125,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/std":8.681012278965849e-05,"train/train/tensor_act_model_layers_23_input_layernorm/std":1.000000072031978,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/std":9.740532838380894e-05,"train/train/tensor_act_model_layers_49_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/mean":-8.440017700195312e-05,"train/train/tensor_act_model_layers_91_self_attn_q_proj/max_abs":7.125,"train/train/tensor_act_model_layers_85_mlp_waleed_W_g/max_abs":3.828125,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/std":3.1439923846144755e-05,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/std":0.0439453125,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/norm":0.019563743265394673,"train/train/layer_model_layers_43/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_mlp_waleed_W_u_weight/norm":4.71875,"train/train/tensor_act_model_layers_15_self_attn/norm":530.6829602877826,"train/train/layer_model_layers_54/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/norm":0.032452628475793284,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/mean":2.1696090698242188e-05,"train/train/tensor_act_model_layers_1_self_attn_q_proj/std":0.8642595776038815,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/mean":-5.066394805908203e-06,"train/train/tensor_act_model_layers_42_self_attn_v_proj/max_abs":3,"train/train/tensor_param_model_layers_80_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_76_mlp_waleed_W_g/max_abs":3.359375,"train/train/tensor_param_model_layers_24_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_28_mlp_down_proj/std":0.059387333363634984,"train/train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_o_proj/max_abs":4.90625,"train/train/tensor_param_model_layers_49_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/norm":0.01947002351003167,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/std":2.691448463301537e-05,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/max_abs":0.125,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/max_abs":0.00121307373046875,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_k_proj/mean":0.028045654296875,"train/train/tensor_act_model_layers_11_self_attn/norm":495.23312504765244,"train/train/tensor_act_model_layers_16_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp_down_proj/norm":1769.409631625082,"train/train/tensor_act_model_layers_0/max_abs":10.9375,"train/train/layer_model_layers_92/grad/mean":-1.3254856031138143e-07,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/mean":-0.00052642822265625,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/mean":-0.00010085105895996094,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp/mean":-0.00675201416015625,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/max_abs":0.27734375,"train/train/tensor_param_model_layers_76_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_2_input_layernorm/norm":5792.603393556225,"train/train/layer_model_layers_63/act/max_abs":24.625,"train/train/tensor_act_model_layers_16_mlp_down_proj/std":0.058166600886850856,"train/train/layer_model_layers_15/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/std":0.00012307636973190057,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/max_abs":0.000102996826171875,"train/train/tensor_act_model_layers_65_mlp_waleed_W_u/max_abs":2.828125,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/norm":0.019047076442488806,"train/train/tensor_param_model_layers_66_mlp_waleed_W_u_weight/max_abs":0.1962890625,"train/train/tensor_param_model_layers_28_mlp_waleed_W_u_weight/mean":-6.437301635742188e-05,"train/train/tensor_param_model_layers_42_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_82_mlp_waleed_W_g_weight/norm":0.028077100655493605,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/mean":0.00014019012451171875,"train/train/tensor_param_model_layers_20_mlp_waleed_W_u_weight/mean":-6.031990051269531e-05,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/std":2.1734329497312236e-05,"train/train/tensor_act_model_layers_44_mlp_waleed/norm":957.9143353620569,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_45_mlp_waleed_W_u/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/mean":-4.7299545258283615e-07,"train/train/layer_model_layers_57/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/mean":-1.1641532182693481e-10,"train/train/tensor_param_model_layers_39_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_mlp_waleed_W_g_weight/norm":4.8125,"train/train/layer_model_layers_16/grad/std":5.1658384780399355e-05,"train/train/tensor_act_model_layers_70_post_attention_layernorm/max_abs":5.1875,"train/train/tensor_param_model_layers_88_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46/frac_near_user_limit":0,"train/train/layer_model_layers_4/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19/std":2.820366701018631,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/max_abs":0.1904296875,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/mean":0.00031280517578125,"train/train/tensor_grad_model_layers_91_mlp_waleed_W_u_weight/mean":-2.3993197828531265e-07,"train/train/layer_model_layers_80/act/norm":24270.925900380134,"train/train/tensor_param_model_layers_44_input_layernorm_weight/max_abs":1,"train/train/layer__model_layers_12/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/mean":-0.00011777877807617188,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_39_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_42_mlp/mean":-4.5027583837509155e-05,"train/train/tensor_act_model_layers_15_mlp_waleed_W_g/std":0.2539062593991938,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/max_abs":0.25,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_k_proj/max_abs":4.3125,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_waleed/frac_near_dtype_limit":0,"train/train/layer__model_layers_52/param/max_abs":1,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/max_abs":0.000270843505859375,"train/train/tensor_grad_model_layers_43_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_69_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/max_abs":0.00156402587890625,"train/train/tensor_param_model_layers_47_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_o_proj/mean":-0.00016647577285766602,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/std":0.029296875,"train/train/tensor_act_model_layers_49_self_attn_o_proj/std":0.03150152124348912,"train/train/layer_model_layers_13/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/norm":0.027253615458838334,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp_waleed_W_g/norm":4653.337474004069,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/max_abs":0.000614166259765625,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/mean":0.00018310546875,"train/train/tensor_grad_model_layers_86_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_post_attention_layernorm/mean":0.0047454833984375,"train/train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_24/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/std":0.0439453125,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/std":0.03466796875,"train/train/tensor_act_model_layers_86_mlp_waleed_W_g/std":0.665041768176669,"train/train/tensor_act_model_layers_29/std":2.699250470441491,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/std":0.0286865234375,"train/train/tensor_grad_model_layers_89_mlp_waleed_W_u_weight/norm":0.034615275136252836,"train/train/tensor_act_model_layers_3_mlp_waleed_W_g/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn/mean":0.020172119140625,"train/train/tensor_act_model_layers_61_post_attention_layernorm/std":1.0000011585937154,"train/train/tensor_act_model_layers_83/max_abs":28.875,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp_waleed_W_u/max_abs":6.5625,"train/train/layer__model_layers_29/param/mean":0.0016876553968408737,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/norm":5.40625,"train/train/tensor_act_model_layers_22_mlp_waleed_W_g/std":0.29541143435546724,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/norm":0.0008771896154939103,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_82/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp_down_proj/norm":352.85763489002466,"train/train/tensor_act_model_layers_17_mlp_waleed/std":0.09655785259165083,"train/train/layer_model_layers_32/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_73_mlp_waleed_W_u_weight/mean":-3.504753112792969e-05,"train/train/tensor_param_model_layers_1_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/mean":-5.364418029785156e-06,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/norm":0.0034216622423630604,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/max_abs":0.248046875,"train/train/tensor_act_model_layers_25_self_attn_v_proj/std":0.35156257883127306,"train/train/tensor_act_model_layers_40_self_attn_q_proj/norm":4479.42914091376,"train/train/tensor_act_model_layers_33_input_layernorm/std":1.0000003762531977,"train/train/tensor_act_model_layers_71_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_o_proj/std":0.23293333455581494,"train/train/tensor_param_model_layers_45_input_layernorm_weight/mean":1,"train/train/layer__model_layers_70/param/std":0.05416899158672965,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/norm":0.03717136225240506,"train/train/tensor_act_model_layers_41_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/max_abs":0.1396484375,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/mean":-6.705522537231445e-06,"train/train/tensor_act_model_layers_92_mlp_waleed_W_u/mean":-0.020233154296875,"train/train/tensor_act_model_layers_20_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/norm":0.006920613644331445,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_31_self_attn_k_proj/std":0.9472677422529332,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/max_abs":0.000568389892578125,"train/train/tensor_act_model_layers_18_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/max_abs":0.2099609375,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/norm":4.6875,"train/train/layer_model_layers_13/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_mlp_waleed_W_g_weight/norm":5.34375,"train/train/layer_model_layers_65/act/mean":0.002865433692932129,"train/train/tensor_act_model_layers_3_post_attention_layernorm/mean":-0.00821685791015625,"train/train/tensor_param_model_layers_39_mlp_waleed_W_u_weight/std":0.025634765625,"train/train/tensor_act_model_layers_46_mlp_waleed_W_u/max_abs":2.328125,"train/train/tensor_act_model_layers_89_mlp_waleed_W_u/max_abs":4.59375,"train/train/tensor_act_model_layers_72/norm":15973.748960772085,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/max_abs":0.00045013427734375,"train/train/tensor_grad_model_layers_44_mlp_waleed_W_u_weight/mean":-1.0221265256404877e-07,"train/train/tensor_param_model_layers_79_mlp_waleed_W_g_weight/max_abs":0.2119140625,"train/train/tensor_act_model_layers_60_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp_waleed/std":0.11694361625642676,"train/train/tensor_act_model_layers_90_mlp_waleed_W_g/max_abs":4.9375,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/max_abs":0.0057373046875,"train/train/tensor_param_model_layers_25_mlp_waleed_W_u_weight/std":0.02392578125,"train/train/tensor_act_model_layers_49_mlp_down_proj/norm":564.1529521831867,"train/train/tensor_act_model_layers_73_self_attn_q_proj/max_abs":5.34375,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/max_abs":0.2294921875,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_mlp_waleed_W_u/norm":2300.1935345957254,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/std":0.00013020905305574732,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/mean":-2.1464074961841106e-09,"train/train/tensor_act_model_layers_24_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/max_abs":0.000240325927734375,"train/train/tensor_act_model_layers_38_post_attention_layernorm/norm":5792.6115722698805,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/std":0.00015284476511289507,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/max_abs":0.000514984130859375,"train/train/layer_model_layers_58/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp_down_proj/max_abs":2.765625,"train/train/tensor_act_model_layers_35_input_layernorm/mean":0.002689361572265625,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_73_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/mean":2.4691689759492874e-07,"train/train/layer_model_layers_6/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/norm":2.84375,"train/train/tensor_act_model_layers_32_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_input_layernorm/max_abs":4.9375,"train/train/tensor_act_model_layers_16_self_attn/std":0.07617740194094678,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/max_abs":0.1689453125,"train/train/tensor_act_model_layers_74_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_waleed_W_u/norm":2493.2292798238736,"train/train/tensor_grad_model_layers_91_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/mean":1.5494879335165024e-07,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/std":0.0235595703125,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/mean":-3.6816345527768135e-08,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_17/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14/norm":17050.05879916185,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/norm":7.34375,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_k_proj/norm":5749.14905651134,"train/train/layer_model_layers_6/act/std":1.10314017356566,"train/train/tensor_act_model_layers_62_post_attention_layernorm/max_abs":4.90625,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/mean":-1.0902993381023407e-05,"train/train/tensor_act_model_layers_56_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_k_proj/max_abs":6.34375,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/norm":0.016655889145769442,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/norm":0.018891835195240923,"train/train/layer__model_layers_15/param/mean":0.0015681262321293632,"train/train/layer_model_layers_16/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/std":4.0759823709903645e-05,"train/train/layer_model_layers_10/act/frac_near_user_limit":0,"train/train/layer__model_layers_55/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn/norm":862.0958548397188,"train/train/tensor_act_model_layers_53_mlp/std":0.0821535980260567,"train/train/tensor_grad_model_layers_37_mlp_waleed_W_u_weight/max_abs":0.000568389892578125,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/mean":-0.000171661376953125,"train/train/layer_model_layers_68/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_mlp_waleed/norm":4190.767156446105,"train/train/tensor_act_model_layers_13_mlp/max_abs":0.57421875,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_93/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_43_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/mean":1.1980533599853516e-05,"train/train/tensor_grad_model_layers_30_mlp_waleed_W_g_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_/mean":2.9225396513938904,"train/train/layer_model_layers_39/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_post_attention_layernorm/std":1.0000004306492865,"train/train/tensor_param_model_layers_65_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_85_mlp_waleed_W_g_weight/norm":0.029640283179436202,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/norm":5.15625,"train/train/tensor_act_model_layers_89_self_attn_v_proj/mean":0.01092529296875,"train/train/tensor_act_model_layers_74_post_attention_layernorm/max_abs":5.40625,"train/train/tensor_act_model_layers_32_self_attn_o_proj/max_abs":0.455078125,"train/train/tensor_act_model_layers_45_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_mlp_waleed_W_g_weight/max_abs":0.140625,"train/train/tensor_act_model_layers_35_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn/max_abs":0.8359375,"train/train/tensor_act_model_layers_68_mlp_down_proj/max_abs":0.90625,"train/train/tensor_grad_model_layers_44_mlp_waleed_W_u_weight/std":3.993707557541207e-05,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/mean":1.5780329704284668e-05,"train/train/tensor_act_model_layers_1_self_attn_k_proj/std":0.9882816891424235,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_input_layernorm/norm":5791.437133791372,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_self_attn/norm":635.5576996376487,"train/train/tensor_act_model_layers_24_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/std":8.512658903065659e-05,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/mean":-1.043517841026187e-07,"train/train/tensor_param_model_layers_69_mlp_waleed_W_u_weight/max_abs":0.1796875,"train/train/tensor_act_model_layers_46_self_attn_v_proj/std":0.3461924612253845,"train/train/tensor_act_model_layers_59_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/max_abs":0.00038909912109375,"train/train/tensor_act_model_layers_55_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/norm":0.020624916336583466,"train/train/tensor_param_model_layers_71_mlp_waleed_W_u_weight/norm":6.21875,"train/train/tensor_act_model_layers_17_mlp_down_proj/mean":0.001007080078125,"train/train/tensor_act_model_layers_56_self_attn_q_proj/norm":5635.161236978225,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/max_abs":0.009033203125,"train/train/tensor_act_model_layers_92_mlp_waleed/mean":0.0010361671447753906,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/norm":5.1875,"train/train/tensor_act_model_layers_34_mlp_waleed/std":0.09045436218162398,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_32_mlp/mean":-0.000606536865234375,"train/train/tensor_grad_model_layers_25_mlp_waleed_W_g_weight/norm":0.015169600181660692,"train/train/tensor_param_model_layers_89_mlp_waleed_W_u_weight/std":0.051025390625,"train/train/tensor_act_model_layers_29/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/max_abs":0.1484375,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/norm":0.010344243611200907,"train/train/tensor_act_model_layers_49_mlp_waleed_W_u/max_abs":2.53125,"train/train/tensor_param_model_layers_24_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16/norm":16871.717613452707,"train/train/tensor_act_model_layers_39_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_33_mlp_waleed_W_g_weight/mean":4.260800778865814e-08,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/max_abs":0.21484375,"train/train/tensor_act_model_layers_36_mlp_down_proj/max_abs":0.54296875,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/mean":8.571147918701172e-05,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/std":4.479448318539793e-05,"train/train/tensor_grad_model_layers_14_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_mlp_waleed_W_u_weight/max_abs":0.29296875,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/mean":-4.013418219983578e-08,"train/train/tensor_act_model_layers_14_mlp/norm":401.4751021816768,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_75/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_77_input_layernorm/mean":0.0051116943359375,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_waleed/std":0.13256905076786685,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/mean":7.455237209796906e-07,"train/train/tensor_act_model_layers_89_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/norm":3.78125,"train/train/tensor_param_model_layers_63_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/std":0.0244140625,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/max_abs":0.1337890625,"train/train/tensor_act_model_layers_43_self_attn_o_proj/norm":671.6598145305732,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/mean":-2.1338462829589844e-05,"train/train/tensor_act_model_layers_30_mlp_waleed_W_u/std":0.2753906356738812,"train/train/tensor_act_model_layers_13_mlp_waleed/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29/max_abs":26.375,"train/train/tensor_param_model_layers_86_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/norm":3.671875,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/std":0.0390625,"train/train/tensor_grad_model_layers_86_mlp_waleed_W_u_weight/mean":3.8975849747657776e-07,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp_waleed_W_g/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_mlp_waleed_W_g_weight/max_abs":0.0006561279296875,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/std":4.411111602035266e-05,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/std":0.052001953125,"train/train/tensor_act_model_layers_55_self_attn_q_proj/norm":5461.277171269169,"train/train/tensor_act_model_layers_36_self_attn_k_proj/norm":5641.834640390842,"train/train/tensor_grad_model_layers_68_mlp_waleed_W_u_weight/norm":0.01999597792117744,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/max_abs":0.000415802001953125,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/mean":0.00010633468627929688,"train/train/tensor_act_model_layers_17_mlp_waleed/norm":790.8383215704639,"train/train/tensor_grad_model_layers_91_mlp_waleed_W_u_weight/max_abs":0.001007080078125,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/max_abs":0.0025787353515625,"train/train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_k_proj/std":1.3281250764341894,"train/train/tensor_act_model_layers_57/std":2.5937750621817637,"train/train/tensor_act_model_layers_33_post_attention_layernorm/mean":0.0014643669128417969,"train/train/layer_model_layers_49/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_mlp_waleed_W_g_weight/norm":4.3125,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_mlp_waleed_W_u_weight/max_abs":0.001007080078125,"train/train/tensor_act_model_layers_58_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_mlp_waleed_W_u_weight/norm":0.02716400468593586,"train/train/tensor_act_model_layers_49/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_34/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/std":0.0003452301025390625,"train/train/tensor_act_model_layers_51_self_attn/norm":1708.5771977640004,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/mean":-1.4371471479535103e-07,"train/train/tensor_act_model_layers_9_mlp_waleed_W_u/norm":2274.5068415575956,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/norm":0.0189277882491083,"train/train/tensor_act_model_layers_17_self_attn_k_proj/std":1.1718751385559556,"train/train/tensor_act_model_layers_47_mlp_down_proj/std":0.06384312567946486,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/mean":0.0001506805419921875,"train/train/tensor_act_model_layers_36_mlp_down_proj/norm":362.3351169526012,"train/train/tensor_act_model_layers_79/std":3.0273524019671445,"train/train/tensor_act_model_layers_9_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/max_abs":2.90625,"train/train/tensor_act_model_layers_61_self_attn_q_proj/max_abs":6.3125,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/std":0.0001024587179484384,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/max_abs":0.154296875,"train/train/tensor_param_model_layers_50_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_9/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp_waleed/mean":0.0035247802734375,"train/train/tensor_act_model_layers_42_mlp_down_proj/std":0.06231699266489778,"train/train/tensor_act_model_layers_64_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_mlp_waleed_W_u_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/max_abs":0.19140625,"train/train/layer_model_layers_47/grad/std":3.32127048644619e-05,"train/train/tensor_act_model_layers_67_mlp_waleed/frac_near_user_limit":0,"train/train/layer_model_layers_67/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp_waleed/max_abs":5.03125,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/mean":4.100799560546875e-05,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_mlp_waleed_W_g_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/mean":-1.2631062418222427e-08,"train/train/tensor_act_model_layers_77_self_attn_k_proj/norm":7078.961298747553,"train/train/tensor_act_model_layers_84_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/mean":-1.0989606380462646e-07,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/mean":0.000339508056640625,"train/train/tensor_act_model_layers_32_self_attn_v_proj/std":0.29541138378032,"train/train/tensor_act_model_layers_72_self_attn_o_proj/norm":1658.9633491716186,"train/train/tensor_grad_model_layers_47_mlp_waleed_W_u_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_16/grad/mean":-1.7182247687055615e-08,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/std":5.8596148714391394e-05,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn/norm":182.37344812187726,"train/train/tensor_param_model_layers_13_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_62_mlp_down_proj/max_abs":0.81640625,"train/train/tensor_grad_model_layers_58_mlp_waleed_W_u_weight/std":4.4891939589181166e-05,"train/train/tensor_act_model_layers_66_self_attn_o_proj/norm":1309.424119383697,"train/train/tensor_act_model_layers_75_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_waleed_W_u/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn/norm":379.34823661869825} \ No newline at end of file diff --git a/wandb/run-20260809_055726-m1dnjnh6/logs/debug-core.log b/wandb/run-20260809_055726-m1dnjnh6/logs/debug-core.log new file mode 100644 index 0000000000000000000000000000000000000000..6ab6c95a5b7718c427082e9937cc663568192adb --- /dev/null +++ b/wandb/run-20260809_055726-m1dnjnh6/logs/debug-core.log @@ -0,0 +1,58 @@ +{"time":"2026-08-09T03:56:53.721154509Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmpyqomfhgp/port-2869678.txt","pid":2869678,"detached":false,"idle-timeout":600000000000,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false} +{"time":"2026-08-09T03:56:53.722258921Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":2869678} +{"time":"2026-08-09T03:56:53.722234577Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-2869678-2909401-2521299324/socket","Net":"unix"}} +{"time":"2026-08-09T03:56:53.900147851Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"} +{"time":"2026-08-09T03:58:19.204299995Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"2(@)"} +{"time":"2026-08-09T03:58:19.282614427Z","level":"INFO","msg":"handleInformInit: received","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T03:58:19.544211049Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T03:58:24.897517362Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"xv8tfjtqu29x"} +{"time":"2026-08-09T05:00:38.67636271Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"xv8tfjtqu29x"} +{"time":"2026-08-09T05:00:40.520576799Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"xv8tfjtqu29x"} +{"time":"2026-08-09T05:00:40.552184373Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T05:00:40.553143275Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608126464Z","level":"INFO","msg":"processOutgoingData: finished","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608106419Z","level":"INFO","msg":"connection: closing","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608216462Z","level":"INFO","msg":"connection: closed successfully","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608222262Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"2(@)"} +{"time":"2026-08-09T05:00:50.241711721Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"3(@)"} +{"time":"2026-08-09T05:00:50.316234404Z","level":"INFO","msg":"handleInformInit: received","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:00:50.57530289Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:00:55.905488223Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"4jxnuihlia82"} +{"time":"2026-08-09T05:57:13.765576248Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"4jxnuihlia82"} +{"time":"2026-08-09T05:57:15.666017984Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"4jxnuihlia82"} +{"time":"2026-08-09T05:57:15.999725293Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:57:16.001240777Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068849513Z","level":"INFO","msg":"connection: closing","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068936758Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068853914Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068948948Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"} +{"time":"2026-08-09T05:57:26.287713024Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"4(@)"} +{"time":"2026-08-09T05:57:26.365087395Z","level":"INFO","msg":"handleInformInit: received","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T05:57:26.623707494Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T05:57:31.962348796Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"rz2ldpq16nht"} +{"time":"2026-08-09T07:02:02.153459338Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"rz2ldpq16nht"} +{"time":"2026-08-09T07:02:04.144735717Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"rz2ldpq16nht"} +{"time":"2026-08-09T07:02:04.180963791Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T07:02:04.181861619Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233097548Z","level":"INFO","msg":"connection: closing","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233183816Z","level":"INFO","msg":"connection: closed successfully","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233105065Z","level":"INFO","msg":"processOutgoingData: finished","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233192773Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"4(@)"} +{"time":"2026-08-09T07:02:13.885910885Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"5(@)"} +{"time":"2026-08-09T07:02:13.954769181Z","level":"INFO","msg":"handleInformInit: received","streamId":"uvqyddz0","id":"5(@)"} +{"time":"2026-08-09T07:02:14.215272022Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"uvqyddz0","id":"5(@)"} +{"time":"2026-08-09T07:02:19.530617395Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"fk71ydt1h8f7"} +{"time":"2026-08-09T07:03:39.9336748Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"fk71ydt1h8f7"} +{"time":"2026-08-09T07:03:40.29901782Z","level":"INFO","msg":"connection: closing","id":"5(@)"} +{"time":"2026-08-09T07:03:40.299111085Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"} +{"time":"2026-08-09T07:03:40.299007456Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"} +{"time":"2026-08-09T07:03:40.299122073Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"} +{"time":"2026-08-09T07:03:42.349833633Z","level":"INFO","msg":"connection: closing","id":"1(@)"} +{"time":"2026-08-09T07:03:42.349942779Z","level":"INFO","msg":"connection: closed successfully","id":"1(@)"} +{"time":"2026-08-09T07:03:42.34985531Z","level":"INFO","msg":"processOutgoingData: finished","id":"1(@)"} +{"time":"2026-08-09T07:03:42.349954994Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"1(@)"} +{"time":"2026-08-09T07:03:42.353940558Z","level":"INFO","msg":"server: parent process exited, terminating service process"} +{"time":"2026-08-09T07:03:42.353988499Z","level":"INFO","msg":"server: is shutting down"} +{"time":"2026-08-09T07:03:42.354128892Z","level":"INFO","msg":"server: listener closed","addr":{"Name":"/tmp/wandb-2869678-2909401-2521299324/socket","Net":"unix"}} +{"time":"2026-08-09T07:03:42.354189401Z","level":"INFO","msg":"server: forced shutdown"} +{"time":"2026-08-09T07:03:42.354198506Z","level":"ERROR","msg":"main: Serve() returned error","error":"forced shutdown"} diff --git a/wandb/run-20260809_055726-m1dnjnh6/logs/debug-internal.log b/wandb/run-20260809_055726-m1dnjnh6/logs/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..cbdc5b1fd1e7ebd814f323ed396a17c14ac2c07b --- /dev/null +++ b/wandb/run-20260809_055726-m1dnjnh6/logs/debug-internal.log @@ -0,0 +1,557 @@ +{"time":"2026-08-09T05:57:26.36521118Z","level":"INFO","msg":"wandb-core"} +{"time":"2026-08-09T05:57:26.365329483Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"} +{"time":"2026-08-09T05:57:26.623538229Z","level":"INFO","msg":"stream: created new stream","id":"m1dnjnh6"} +{"time":"2026-08-09T05:57:26.623615023Z","level":"INFO","msg":"handler: started"} +{"time":"2026-08-09T05:57:26.623693126Z","level":"INFO","msg":"stream: started"} +{"time":"2026-08-09T05:57:26.623702658Z","level":"INFO","msg":"writer: started","stream_id":"m1dnjnh6"} +{"time":"2026-08-09T05:57:26.62372587Z","level":"INFO","msg":"sender: started"} +{"time":"2026-08-09T05:57:27.557642621Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1} +{"time":"2026-08-09T05:57:27.650541559Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:57:42.558431814Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":0,"events_lines":2,"console_offset":1,"console_lines":4,"uploaded_len":2} +{"time":"2026-08-09T05:57:42.665431125Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:57:57.558178253Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":2,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:57:57.66806981Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:58:12.558545553Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":4,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:58:12.675065428Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:58:25.77940765Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":1201} +{"time":"2026-08-09T05:58:25.779432289Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1} +{"time":"2026-08-09T05:58:25.784311509Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":2943} +{"time":"2026-08-09T05:58:25.786109454Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":568} +{"time":"2026-08-09T05:58:25.795605792Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":6377} +{"time":"2026-08-09T05:58:25.799631102Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":929} +{"time":"2026-08-09T05:58:25.801373004Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":7850} +{"time":"2026-08-09T05:58:25.801400138Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1} +{"time":"2026-08-09T05:58:25.8077892Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":10388} +{"time":"2026-08-09T05:58:25.807878363Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":2} +{"time":"2026-08-09T05:58:25.809530872Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":11050} +{"time":"2026-08-09T05:58:25.809623136Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":11} +{"time":"2026-08-09T05:58:25.809939148Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":11204} +{"time":"2026-08-09T05:58:25.810362455Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":114} +{"time":"2026-08-09T05:58:25.8129337Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":12370} +{"time":"2026-08-09T05:58:25.81302688Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":8} +{"time":"2026-08-09T05:58:25.820493164Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":14756} +{"time":"2026-08-09T05:58:25.820610926Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":4} +{"time":"2026-08-09T05:58:25.82072297Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":14797} +{"time":"2026-08-09T05:58:25.820856253Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":21} +{"time":"2026-08-09T05:58:25.825688099Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":16586} +{"time":"2026-08-09T05:58:25.825776651Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":3} +{"time":"2026-08-09T05:58:25.83122853Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":18671} +{"time":"2026-08-09T05:58:25.833046941Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":643} +{"time":"2026-08-09T05:58:27.603035488Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":1,"events_offset":6,"events_lines":2,"console_offset":4,"console_lines":2} +{"time":"2026-08-09T05:58:28.695383891Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:58:42.559434891Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":8,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:58:42.665554062Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:58:57.558707275Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":10,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:58:57.659527523Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:59:12.557745535Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":12,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:59:12.664376015Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:59:27.581994176Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":1,"history_lines":1,"events_offset":14,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:59:28.624785257Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:59:42.557946915Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":16,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:59:42.686250668Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T05:59:57.558544029Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":18,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T05:59:57.667102572Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:00:12.580541104Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":2,"history_lines":1,"events_offset":20,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T06:00:13.598640632Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:00:27.55855345Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":22,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T06:00:27.657705204Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:00:42.575471197Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":3,"history_lines":1,"events_offset":24,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T06:00:43.727701168Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:00:57.557878271Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":26,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T06:00:57.673021214Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:01:12.558549696Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":28,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T06:01:12.672934772Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:01:27.578920263Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":4,"history_lines":1,"events_offset":30,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T06:01:28.588451091Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:01:42.558167743Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":32,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T06:01:42.667970439Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:01:57.558214304Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":34,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T06:01:57.655987472Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:02:12.579306836Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":5,"history_lines":1,"events_offset":36,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T06:02:13.713510257Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:02:27.558083823Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":38,"events_lines":2,"console_offset":6,"console_lines":6} +{"time":"2026-08-09T06:02:27.664788692Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:02:42.578126148Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":6,"history_lines":1,"events_offset":40,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T06:02:43.638509285Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:02:57.558581737Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":42,"events_lines":2,"console_offset":7,"console_lines":1} +{"time":"2026-08-09T06:02:57.667861039Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:03:12.558234975Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":44,"events_lines":2,"console_offset":12,"console_lines":5} +{"time":"2026-08-09T06:03:12.659155104Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:03:27.581373227Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":7,"history_lines":1,"events_offset":46,"events_lines":2,"console_offset":16,"console_lines":2} +{"time":"2026-08-09T06:03:28.610415256Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:03:42.558040291Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":48,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:03:42.670277722Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:03:57.558363002Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":50,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:03:57.663129245Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:04:12.558553586Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":52,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:04:12.658259898Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:04:27.578716295Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":8,"history_lines":1,"events_offset":54,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:04:28.574645711Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:04:42.558485934Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":56,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:04:42.685014412Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:04:57.557976397Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":58,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:04:57.67470054Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:05:12.57887198Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":9,"history_lines":1,"events_offset":60,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:05:13.638692787Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:05:27.558059748Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":62,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:05:27.681436405Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:05:42.576046816Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":10,"history_lines":1,"events_offset":64,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:05:43.684920637Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:05:57.557919145Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":66,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:05:57.665279762Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:06:12.558397759Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":68,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:06:12.659761511Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:06:27.578525066Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":11,"history_lines":1,"events_offset":70,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:06:28.580988803Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:06:42.557880591Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":72,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:06:42.65617582Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:06:57.557729747Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":74,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:06:57.678050279Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:07:12.575826026Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":12,"history_lines":1,"events_offset":76,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:07:13.548095095Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:07:27.557840231Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":78,"events_lines":2,"console_offset":18,"console_lines":6} +{"time":"2026-08-09T06:07:27.660622951Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:07:42.585091739Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":13,"history_lines":1,"events_offset":80,"events_lines":2,"console_offset":16,"console_lines":1} +{"time":"2026-08-09T06:07:43.665714599Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:07:57.558241994Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":82,"events_lines":2,"console_offset":19,"console_lines":1} +{"time":"2026-08-09T06:07:57.641080213Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:08:12.558465684Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":84,"events_lines":2,"console_offset":24,"console_lines":5} +{"time":"2026-08-09T06:08:12.679235287Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:08:27.577006899Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":14,"history_lines":1,"events_offset":86,"events_lines":2,"console_offset":28,"console_lines":2} +{"time":"2026-08-09T06:08:28.591603191Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:08:42.557684981Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":88,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:08:42.662500579Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:08:57.558417339Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":90,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:08:57.656450489Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:09:12.557864985Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":92,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:09:12.649202135Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:09:27.575542742Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":15,"history_lines":1,"events_offset":94,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:09:28.543210733Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:09:42.557960122Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":96,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:09:42.662583569Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:09:57.558357536Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":98,"events_lines":2,"console_offset":30,"console_lines":2} +{"time":"2026-08-09T06:09:57.680989785Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:10:12.587309713Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":16,"history_lines":1,"events_offset":100,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:10:13.640640202Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:10:27.558326471Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":102,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:10:27.671292602Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:10:42.58048586Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":17,"history_lines":1,"events_offset":104,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:10:43.577132018Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:10:57.558429898Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":106,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:10:57.695936575Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:11:12.558748621Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":108,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:11:12.667518264Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:11:27.578765719Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":18,"history_lines":1,"events_offset":110,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:11:28.580405591Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:11:42.558602966Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":112,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:11:42.68333918Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:11:57.558181335Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":114,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:11:57.668396382Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:12:12.576711012Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":19,"history_lines":1,"events_offset":116,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:12:13.677729904Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:12:27.557910349Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":118,"events_lines":2,"console_offset":31,"console_lines":5} +{"time":"2026-08-09T06:12:27.659120789Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:12:42.577351419Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":20,"history_lines":1,"events_offset":120,"events_lines":2,"console_offset":28,"console_lines":1} +{"time":"2026-08-09T06:12:43.639288725Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:12:57.558290044Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":122,"events_lines":2,"console_offset":31,"console_lines":1} +{"time":"2026-08-09T06:12:57.664951611Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:13:12.558316572Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":124,"events_lines":2,"console_offset":36,"console_lines":5} +{"time":"2026-08-09T06:13:12.660996321Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:13:27.57859599Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":21,"history_lines":1,"events_offset":126,"events_lines":2,"console_offset":40,"console_lines":2} +{"time":"2026-08-09T06:13:28.601899488Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:13:42.558565062Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":128,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:13:42.652034494Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:13:57.558757776Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":130,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:13:57.645338884Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:14:12.558665994Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":132,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:14:12.662662346Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:14:27.574702831Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":22,"history_lines":1,"events_offset":134,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:14:28.691839205Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:14:42.558232951Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":136,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:14:42.660763025Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:14:57.557915305Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":138,"events_lines":2,"console_offset":42,"console_lines":2} +{"time":"2026-08-09T06:14:57.688855527Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:15:12.580627155Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":23,"history_lines":1,"events_offset":140,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:15:13.661390338Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:15:27.557991848Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":142,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:15:27.667894012Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:15:42.578470892Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":24,"history_lines":1,"events_offset":144,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:15:43.586523614Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:15:57.5579984Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":146,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:15:57.666505483Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:16:12.557867803Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":148,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:16:12.674199804Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:16:27.57488781Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":25,"history_lines":1,"events_offset":150,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:16:28.543436071Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:16:42.557847096Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":152,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:16:42.674971017Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:16:57.558542895Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":154,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:16:57.652959457Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:17:12.579790136Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":26,"history_lines":1,"events_offset":156,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:17:13.734055102Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:17:27.558543131Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":158,"events_lines":2,"console_offset":43,"console_lines":5} +{"time":"2026-08-09T06:17:27.661211819Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:17:42.580438968Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":27,"history_lines":1,"events_offset":160,"events_lines":2,"console_offset":40,"console_lines":1} +{"time":"2026-08-09T06:17:43.580515078Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:17:57.557900777Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":162,"events_lines":2,"console_offset":43,"console_lines":1} +{"time":"2026-08-09T06:17:57.665075773Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:18:12.558340543Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":164,"events_lines":2,"console_offset":48,"console_lines":5} +{"time":"2026-08-09T06:18:12.649854431Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:18:27.576706705Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":28,"history_lines":1,"events_offset":166,"events_lines":2,"console_offset":52,"console_lines":2} +{"time":"2026-08-09T06:18:28.678077696Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:18:42.558390215Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":168,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:18:42.677059531Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:18:57.55846931Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":170,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:18:57.65213461Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:19:12.558678765Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":172,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:19:12.697411918Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:19:27.580444288Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":29,"history_lines":1,"events_offset":174,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:19:28.624993265Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:19:42.558498956Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":176,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:19:42.659164441Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:19:57.574444664Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":30,"history_lines":1,"events_offset":178,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:19:58.586511687Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:20:12.558666623Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":180,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:20:12.654137835Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:20:27.577018252Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":31,"history_lines":1,"events_offset":182,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:20:28.619521883Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:20:42.557923911Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":184,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:20:42.645546734Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:20:57.558692489Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":186,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:20:57.668422523Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:21:12.575242575Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":32,"history_lines":1,"events_offset":188,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:21:13.584105457Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:21:27.558782184Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":190,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:21:27.666106619Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:21:42.558034764Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":192,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:21:42.666481024Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:21:57.581238034Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":33,"history_lines":1,"events_offset":194,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:21:58.45996068Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:22:12.557759514Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":196,"events_lines":2,"console_offset":54,"console_lines":6} +{"time":"2026-08-09T06:22:12.659959945Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:22:27.584835322Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":34,"history_lines":1,"events_offset":198,"events_lines":2,"console_offset":52,"console_lines":1} +{"time":"2026-08-09T06:22:28.520336942Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:22:42.558233291Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":200,"events_lines":2,"console_offset":55,"console_lines":1} +{"time":"2026-08-09T06:22:42.675080223Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:22:57.558567426Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":202,"events_lines":2,"console_offset":60,"console_lines":7} +{"time":"2026-08-09T06:22:57.661189864Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:23:12.600436434Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":35,"history_lines":1,"events_offset":204,"events_lines":2,"console_offset":66,"console_lines":2} +{"time":"2026-08-09T06:23:13.754028152Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:23:27.558413173Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":206,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T06:23:27.649020741Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:23:42.578822645Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":36,"history_lines":1,"events_offset":208,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T06:23:43.650889822Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:23:57.558605554Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":210,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T06:23:57.658048396Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:24:12.558213546Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":212,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T06:24:12.661302405Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:24:27.574601668Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":37,"history_lines":1,"events_offset":214,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T06:24:28.588680342Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:24:42.579731459Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":38,"history_lines":1,"events_offset":216,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T06:24:43.713682462Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:24:57.558024726Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":218,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T06:24:57.661226678Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:25:12.55781347Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":220,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T06:25:12.682052902Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:25:27.579276391Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":39,"history_lines":1,"events_offset":222,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T06:25:28.637872539Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:25:42.557934945Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":224,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T06:25:42.654359734Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:25:57.57508167Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":40,"history_lines":1,"events_offset":226,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T06:25:58.52902003Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:26:12.579996166Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":41,"history_lines":1,"events_offset":228,"events_lines":2,"console_offset":66,"console_lines":1} +{"time":"2026-08-09T06:26:13.743689299Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:26:27.558737178Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":230,"events_lines":2,"console_offset":68,"console_lines":11} +{"time":"2026-08-09T06:26:27.664368927Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:26:42.55876572Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":232,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T06:26:42.659625704Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:26:57.577159872Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":42,"history_lines":1,"events_offset":234,"events_lines":2,"console_offset":78,"console_lines":2} +{"time":"2026-08-09T06:26:58.542198176Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:27:12.558164998Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":236,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T06:27:12.667463926Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:27:27.557763866Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":238,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T06:27:27.653123402Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:27:42.57614813Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":43,"history_lines":1,"events_offset":240,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T06:27:43.602213302Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:27:57.557827205Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":242,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T06:27:57.660489148Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:28:12.577542096Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":44,"history_lines":1,"events_offset":244,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T06:28:13.577369525Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:28:27.558221384Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":246,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T06:28:27.67619672Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:28:42.58255326Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":45,"history_lines":1,"events_offset":248,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T06:28:43.659110407Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:28:57.557866689Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":250,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T06:28:57.662452686Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:29:12.577566642Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":46,"history_lines":1,"events_offset":252,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T06:29:13.621740942Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:29:27.558482793Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":254,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T06:29:27.659309666Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:29:42.557974748Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":256,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T06:29:42.660016584Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:29:57.577395499Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":47,"history_lines":1,"events_offset":258,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T06:29:58.55596885Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:30:12.57497553Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":48,"history_lines":1,"events_offset":260,"events_lines":2,"console_offset":78,"console_lines":1} +{"time":"2026-08-09T06:30:13.639875199Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:30:27.558542608Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":262,"events_lines":2,"console_offset":80,"console_lines":11} +{"time":"2026-08-09T06:30:27.670657068Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:30:42.558483455Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":264,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T06:30:42.661722191Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:30:57.581234177Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":49,"history_lines":1,"events_offset":266,"events_lines":2,"console_offset":90,"console_lines":2} +{"time":"2026-08-09T06:30:58.575236367Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:31:12.558063685Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":268,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T06:31:12.677672989Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:31:27.557761201Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":270,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T06:31:27.650073223Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:31:42.577064086Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":50,"history_lines":1,"events_offset":272,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T06:31:43.659206773Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:31:57.55786681Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":274,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T06:31:57.69026715Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:32:12.578575875Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":51,"history_lines":1,"events_offset":276,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T06:32:13.551087789Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:32:27.575796882Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":52,"history_lines":1,"events_offset":278,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T06:32:28.546539363Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:32:42.55800477Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":280,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T06:32:42.665972067Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:32:57.557717923Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":282,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T06:32:57.681643813Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:33:12.573514931Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":53,"history_lines":1,"events_offset":284,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T06:33:13.546953994Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:33:27.558284273Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":286,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T06:33:27.65888269Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:33:42.558476138Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":288,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T06:33:42.661755906Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:33:57.582764216Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":54,"history_lines":1,"events_offset":290,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T06:33:58.563761609Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:34:12.574736072Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":55,"history_lines":1,"events_offset":292,"events_lines":2,"console_offset":90,"console_lines":1} +{"time":"2026-08-09T06:34:13.511386896Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:34:27.558497244Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":294,"events_lines":2,"console_offset":92,"console_lines":11} +{"time":"2026-08-09T06:34:27.646540143Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:34:42.55772009Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":296,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T06:34:42.681537208Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:34:57.577957014Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":56,"history_lines":1,"events_offset":298,"events_lines":2,"console_offset":102,"console_lines":2} +{"time":"2026-08-09T06:34:58.622234268Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:35:12.558469912Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":300,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T06:35:12.665339193Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:35:27.576933814Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":57,"history_lines":1,"events_offset":302,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T06:35:28.574711287Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:35:42.557947812Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":304,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T06:35:42.666933749Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:35:57.558699703Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":306,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T06:35:57.674133239Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:36:12.578470165Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":58,"history_lines":1,"events_offset":308,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T06:36:13.932384652Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:36:27.57572659Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":59,"history_lines":1,"events_offset":310,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T06:36:28.494915371Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:36:42.558014413Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":312,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T06:36:42.641978914Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:36:57.558674545Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":314,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T06:36:57.665085329Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:37:12.580256972Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":60,"history_lines":1,"events_offset":316,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T06:37:13.6180239Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:37:27.558247761Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":318,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T06:37:27.642652932Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:37:42.558149477Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":320,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T06:37:42.657861935Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:37:57.574103541Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":61,"history_lines":1,"events_offset":322,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T06:37:58.552630435Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:38:12.576805071Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":62,"history_lines":1,"events_offset":324,"events_lines":2,"console_offset":102,"console_lines":1} +{"time":"2026-08-09T06:38:13.600937965Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:38:27.558168048Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":326,"events_lines":2,"console_offset":104,"console_lines":11} +{"time":"2026-08-09T06:38:27.670905759Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:38:42.575187134Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":63,"history_lines":1,"events_offset":328,"events_lines":2,"console_offset":114,"console_lines":2} +{"time":"2026-08-09T06:38:43.714252373Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:38:57.557868647Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":330,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T06:38:57.670135892Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:39:12.55853622Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":332,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T06:39:12.661725042Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:39:27.574211624Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":64,"history_lines":1,"events_offset":334,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T06:39:28.656601352Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:39:42.558389592Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":336,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T06:39:42.667181754Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:39:57.558112669Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":338,"events_lines":2,"console_offset":116,"console_lines":2} +{"time":"2026-08-09T06:39:57.654736433Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:40:12.578044337Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":65,"history_lines":1,"events_offset":340,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T06:40:13.642179669Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:40:27.575550155Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":66,"history_lines":1,"events_offset":342,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T06:40:28.588134017Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:40:42.558672811Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":344,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T06:40:42.677950337Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:40:57.558480779Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":346,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T06:40:57.639082801Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:41:12.575636941Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":67,"history_lines":1,"events_offset":348,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T06:41:13.572169408Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:41:27.558669849Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":350,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T06:41:27.664915087Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:41:42.585551678Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":68,"history_lines":1,"events_offset":352,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T06:41:43.513303346Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:41:57.575238547Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":69,"history_lines":1,"events_offset":354,"events_lines":2,"console_offset":114,"console_lines":1} +{"time":"2026-08-09T06:41:58.645950353Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:42:12.558205454Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":356,"events_lines":2,"console_offset":117,"console_lines":12} +{"time":"2026-08-09T06:42:12.66562487Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:42:27.55833331Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":358,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T06:42:27.676622541Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:42:42.596692429Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":70,"history_lines":1,"events_offset":360,"events_lines":2,"console_offset":128,"console_lines":2} +{"time":"2026-08-09T06:42:43.829534626Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:42:57.55783239Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":362,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T06:42:57.66302413Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:43:12.558630816Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":364,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T06:43:12.64991426Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:43:27.575744703Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":71,"history_lines":1,"events_offset":366,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T06:43:28.573691104Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:43:42.55804968Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":368,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T06:43:42.655593153Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:43:57.558282793Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":370,"events_lines":2,"console_offset":130,"console_lines":2} +{"time":"2026-08-09T06:43:57.668872739Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:44:12.582768362Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":72,"history_lines":1,"events_offset":372,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T06:44:13.575799055Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:44:27.57560748Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":73,"history_lines":1,"events_offset":374,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T06:44:28.747341938Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:44:42.558614293Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":376,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T06:44:42.661736935Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:44:57.557999402Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":378,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T06:44:57.672150345Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:45:12.583214958Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":74,"history_lines":1,"events_offset":380,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T06:45:13.665958762Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:45:27.558118301Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":382,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T06:45:27.64380403Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:45:42.578936716Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":75,"history_lines":1,"events_offset":384,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T06:45:43.600210294Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:45:57.586030473Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":76,"history_lines":1,"events_offset":386,"events_lines":2,"console_offset":128,"console_lines":1} +{"time":"2026-08-09T06:45:58.666410805Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:46:12.55849205Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":388,"events_lines":2,"console_offset":131,"console_lines":10} +{"time":"2026-08-09T06:46:12.648530707Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:46:27.55842782Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":390,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T06:46:27.66338294Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:46:42.574949052Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":77,"history_lines":1,"events_offset":392,"events_lines":2,"console_offset":140,"console_lines":2} +{"time":"2026-08-09T06:46:43.666389125Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:46:57.557909998Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":394,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T06:46:57.654965578Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:47:12.55818156Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":396,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T06:47:12.668897077Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:47:27.577308199Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":78,"history_lines":1,"events_offset":398,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T06:47:28.689507802Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:47:42.558157526Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":400,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T06:47:42.650477575Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:47:57.575853478Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":79,"history_lines":1,"events_offset":402,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T06:47:58.559633495Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:48:12.55823923Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":404,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T06:48:12.672799433Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:48:27.581193875Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":80,"history_lines":1,"events_offset":406,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T06:48:28.517961299Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:48:42.558102284Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":408,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T06:48:42.696565799Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:48:57.577979166Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":81,"history_lines":1,"events_offset":410,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T06:48:58.515979732Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:49:12.558493249Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":412,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T06:49:12.675770537Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:49:27.558305294Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":414,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T06:49:27.665820504Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:49:42.573473214Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":82,"history_lines":1,"events_offset":416,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T06:49:43.676715587Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:49:57.575250244Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":83,"history_lines":1,"events_offset":418,"events_lines":2,"console_offset":140,"console_lines":1} +{"time":"2026-08-09T06:49:58.554224156Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:50:12.558421479Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":420,"events_lines":2,"console_offset":142,"console_lines":11} +{"time":"2026-08-09T06:50:12.659827488Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:50:27.558718412Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":422,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T06:50:27.661878488Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:50:42.574192759Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":84,"history_lines":1,"events_offset":424,"events_lines":2,"console_offset":152,"console_lines":2} +{"time":"2026-08-09T06:50:43.580312315Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:50:57.558505507Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":426,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T06:50:57.668529845Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:51:12.557939661Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":428,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T06:51:12.65056213Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:51:27.578323861Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":85,"history_lines":1,"events_offset":430,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T06:51:28.636333625Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:51:42.558311051Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":432,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T06:51:42.665390933Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:51:57.575385273Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":86,"history_lines":1,"events_offset":434,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T06:51:58.646417508Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:52:12.583538473Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":87,"history_lines":1,"events_offset":436,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T06:52:13.559436733Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:52:27.558139769Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":438,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T06:52:27.658241524Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:52:42.558528262Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":440,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T06:52:42.662055638Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:52:57.576442171Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":88,"history_lines":1,"events_offset":442,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T06:52:58.626682855Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:53:12.558431194Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":444,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T06:53:12.666585781Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:53:27.558624714Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":446,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T06:53:27.650321415Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:53:42.577990956Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":89,"history_lines":1,"events_offset":448,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T06:53:43.508093745Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:53:57.584261945Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":90,"history_lines":1,"events_offset":450,"events_lines":2,"console_offset":152,"console_lines":1} +{"time":"2026-08-09T06:53:58.557079458Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:54:12.558643293Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":452,"events_lines":2,"console_offset":154,"console_lines":11} +{"time":"2026-08-09T06:54:12.658875423Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:54:27.558560272Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":454,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T06:54:27.670633256Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:54:42.575148415Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":91,"history_lines":1,"events_offset":456,"events_lines":2,"console_offset":164,"console_lines":2} +{"time":"2026-08-09T06:54:43.71772473Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:54:57.558500851Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":458,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T06:54:57.662982538Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:55:12.583090921Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":92,"history_lines":1,"events_offset":460,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T06:55:13.589840247Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:55:27.558138847Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":462,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T06:55:27.649709123Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:55:42.558538359Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":464,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T06:55:42.661887786Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:55:57.5787472Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":93,"history_lines":1,"events_offset":466,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T06:55:58.611121143Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:56:12.575672773Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":94,"history_lines":1,"events_offset":468,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T06:56:13.602486252Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:56:27.558391748Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":470,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T06:56:27.66631348Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:56:42.558122196Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":472,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T06:56:42.664320492Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:56:57.577153939Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":95,"history_lines":1,"events_offset":474,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T06:56:58.559072703Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:57:12.558405736Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":476,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T06:57:12.659404639Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:57:27.573778978Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":96,"history_lines":1,"events_offset":478,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T06:57:28.538011266Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:57:42.557751403Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":480,"events_lines":2,"console_offset":166,"console_lines":6} +{"time":"2026-08-09T06:57:42.806919018Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:57:57.586191157Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":97,"history_lines":1,"events_offset":482,"events_lines":2,"console_offset":164,"console_lines":1} +{"time":"2026-08-09T06:57:58.549970155Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:58:12.558364364Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":484,"events_lines":2,"console_offset":167,"console_lines":1} +{"time":"2026-08-09T06:58:12.6415133Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:58:27.578488655Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":98,"history_lines":1,"events_offset":486,"events_lines":2,"console_offset":172,"console_lines":6} +{"time":"2026-08-09T06:58:28.596275921Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:58:42.558441546Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":488,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T06:58:42.680367181Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:58:57.558776052Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":490,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T06:58:57.655394657Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:59:12.576348928Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":99,"history_lines":1,"events_offset":492,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T06:59:13.624944148Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:59:27.558343429Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":494,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T06:59:27.660370929Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:59:42.600721158Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":100,"history_lines":1,"events_offset":496,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T06:59:43.847833547Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:59:57.558224759Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":498,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T06:59:57.659180582Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:00:12.578096433Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":101,"history_lines":1,"events_offset":500,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T07:00:13.552399503Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:00:27.558332471Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":502,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T07:00:27.666356289Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:00:42.558343123Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":504,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T07:00:42.66135776Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:00:57.573837994Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":102,"history_lines":1,"events_offset":506,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T07:00:58.952626347Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:01:12.557877794Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":508,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T07:01:12.679879984Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:01:27.576152305Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":103,"history_lines":1,"events_offset":510,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T07:01:28.652668151Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:01:42.57865814Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":104,"history_lines":2,"events_offset":512,"events_lines":2,"console_offset":176,"console_lines":1} +{"time":"2026-08-09T07:01:43.533791887Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:01:57.558121718Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":514,"events_lines":2,"console_offset":178,"console_lines":13} +{"time":"2026-08-09T07:01:57.68063451Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:02:02.972552361Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"} +{"time":"2026-08-09T07:02:02.994216029Z","level":"INFO","msg":"filestream: sending request","total_files":3,"history_offset":106,"history_lines":1,"console_offset":190,"console_lines":31,"uploaded_len":3,"complete":true,"exit_code":0} +{"time":"2026-08-09T07:02:04.117544434Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:02:04.118942009Z","level":"INFO","msg":"handler: operation stats","stats":{}} +{"time":"2026-08-09T07:02:04.181011415Z","level":"INFO","msg":"stream: finishing up"} +{"time":"2026-08-09T07:02:04.181046934Z","level":"INFO","msg":"handler: closed"} +{"time":"2026-08-09T07:02:04.181165866Z","level":"INFO","msg":"sender: closed"} +{"time":"2026-08-09T07:02:04.181169919Z","level":"INFO","msg":"stream: all finished"} diff --git a/wandb/run-20260809_055726-m1dnjnh6/logs/debug.log b/wandb/run-20260809_055726-m1dnjnh6/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..f7db0b33ebb36ed290d47d7a9630fa9665fed465 --- /dev/null +++ b/wandb/run-20260809_055726-m1dnjnh6/logs/debug.log @@ -0,0 +1,28 @@ +2026-08-09 05:57:26,363 INFO MainThread:4056178 [wandb_setup.py:_flush():81] Current SDK version is 0.28.1 +2026-08-09 05:57:26,363 INFO MainThread:4056178 [wandb_setup.py:_flush():81] Configure stats pid to 4056178 +2026-08-09 05:57:26,363 INFO MainThread:4056178 [wandb_setup.py:_flush():81] Loading settings from environment variables +2026-08-09 05:57:26,363 INFO MainThread:4056178 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_055726-m1dnjnh6/logs/debug.log +2026-08-09 05:57:26,363 INFO MainThread:4056178 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_055726-m1dnjnh6/logs/debug-internal.log +2026-08-09 05:57:26,363 INFO MainThread:4056178 [wandb_init.py:init():772] calling init triggers +2026-08-09 05:57:26,363 INFO MainThread:4056178 [wandb_init.py:init():777] wandb.init called with sweep_config: {} +config: {'_wandb': {}} +2026-08-09 05:57:26,363 INFO MainThread:4056178 [wandb_init.py:init():820] starting backend +2026-08-09 05:57:26,363 INFO MainThread:4056178 [wandb_init.py:init():826] Connected to an existing wandb-core service via WANDB_SERVICE +2026-08-09 05:57:26,364 INFO MainThread:4056178 [wandb_init.py:init():835] sending inform_init request +2026-08-09 05:57:26,624 INFO MainThread:4056178 [wandb_init.py:init():840] backend started and connected +2026-08-09 05:57:26,627 INFO MainThread:4056178 [wandb_init.py:init():910] updated telemetry +2026-08-09 05:57:26,634 INFO MainThread:4056178 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout +2026-08-09 05:57:26,882 INFO MainThread:4056178 [wandb_init.py:init():978] starting run threads in backend +2026-08-09 05:57:26,957 INFO MainThread:4056178 [wandb_run.py:_console_start():2621] atexit reg +2026-08-09 05:57:26,958 INFO MainThread:4056178 [wandb_run.py:_redirect():2471] redirect: wrap_raw +2026-08-09 05:57:26,958 INFO MainThread:4056178 [wandb_run.py:_redirect():2540] Wrapping output streams. +2026-08-09 05:57:26,958 INFO MainThread:4056178 [wandb_run.py:_redirect():2563] Redirects installed. +2026-08-09 05:57:26,960 INFO MainThread:4056178 [wandb_init.py:init():1016] run started, returning control to user process +2026-08-09 05:57:26,961 INFO MainThread:4056178 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.15.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 94, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'glu', 'activation': 'waleed', 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/glu-waleed-94L_run', 'per_device_train_batch_size': 128, 'num_train_epochs': 1, 'max_steps': 1500, 'learning_rate': 0.001, 'lr_scheduler_type': 'constant', 'lr_scheduler_kwargs': None, 'warmup_steps': 0, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.01, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 4, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-glu-waleed-94L-15.9M-20260809-055724', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 50, 'eval_delay': 0, 'per_device_eval_batch_size': 128, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': True, 'hub_token': '', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/A-glu-waleed-94L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1} +2026-08-09 05:57:26,965 INFO MainThread:4056178 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 15949440 - > +2026-08-09 05:57:26,965 INFO MainThread:4056178 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 15949440 None +2026-08-09 07:02:02,152 INFO MainThread:4056178 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/huggingface/m1dnjnh6 +2026-08-09 07:02:02,153 INFO MainThread:4056178 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0 +2026-08-09 07:02:02,153 INFO MainThread:4056178 [wandb_run.py:_restore():2570] restore +2026-08-09 07:02:02,153 INFO MainThread:4056178 [wandb_run.py:_restore():2576] restore done +2026-08-09 07:02:04,180 INFO MainThread:4056178 [wandb_run.py:_footer_sync_info():3993] logging synced files diff --git a/wandb/run-20260809_055726-m1dnjnh6/run-m1dnjnh6.wandb b/wandb/run-20260809_055726-m1dnjnh6/run-m1dnjnh6.wandb new file mode 100644 index 0000000000000000000000000000000000000000..ce2b6a27d551ec04795526c5bd534f3d66d169b8 --- /dev/null +++ b/wandb/run-20260809_055726-m1dnjnh6/run-m1dnjnh6.wandb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:355ff324c4a1913b12edd4ede026bc662bc410e6fe34cf3123e6e7e573633bdb +size 19128467 diff --git a/wandb/run-20260809_061951-oe9tdw54/files/config.yaml b/wandb/run-20260809_061951-oe9tdw54/files/config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..784be3d3a841a6aff1ad36c469de0cec46eb7b03 --- /dev/null +++ b/wandb/run-20260809_061951-oe9tdw54/files/config.yaml @@ -0,0 +1,432 @@ +_name_or_path: + value: "" +_wandb: + value: + cli_version: 0.28.1 + e: + m9tdjbfp3yvliqsqunb1cen1ueg4937y: + args: + - --config + - /mnt/data/zainulabideen/zain-exp/notebooks/Activation/configs/baseline1.yaml + - --variants + - mlp-tanh-9L + - --push + codePath: sweep.py + codePathLocal: sweep.py + cpu_count: 112 + cpu_count_logical: 224 + cudaVersion: "12.4" + disk: + /: + total: "1560765693952" + used: "708290355200" + email: deepnevro@gmail.com + executable: /mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python + git: + commit: 34b8d2e8f9a0c5751333310e69fa0c1056381deb + remote: https://github.com/deepnevro/Activation.git + gpu: NVIDIA H100 80GB HBM3 + gpu_count: 8 + gpu_nvidia: + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-39c684a5-fde6-83d7-1663-0859795881ae + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-132944c4-b689-2b5f-89a4-d730401677ab + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bc6c3e3c-9b90-09ca-c034-774961847c54 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea + host: deeplens-k3s-node1 + memory: + total: "2164089937920" + os: Linux-5.15.0-126-generic-x86_64-with-glibc2.35 + program: /mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py + python: CPython 3.11.15 + root: /mnt/data/zainulabideen/zain-exp/notebooks/Activation + startedAt: "2026-08-09T06:19:51.990723Z" + writerId: m9tdjbfp3yvliqsqunb1cen1ueg4937y + m: + - "1": train/global_step + "6": + - 3 + "7": [] + - "2": '*' + "5": 1 + "6": + - 1 + "7": [] + python_version: 3.11.15 + t: + "1": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "2": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "3": + - 2 + - 7 + - 13 + - 19 + - 41 + - 66 + "4": 3.11.15 + "5": 0.28.1 + "6": 5.15.0.dev0 + "9": + "1": transformers_trainer + "12": 0.28.1 + "13": linux-x86_64 +accelerator_config: + value: + dispatch_batches: null + even_batches: true + gradient_accumulation_kwargs: null + non_blocking: false + split_batches: false + use_seedable_sampler: true +activation: + value: tanh +adam_beta1: + value: 0.9 +adam_beta2: + value: 0.999 +adam_epsilon: + value: 1e-08 +architectures: + value: null +attention_bias: + value: false +attention_dropout: + value: 0 +auto_find_batch_size: + value: false +average_tokens_across_devices: + value: true +batch_eval_metrics: + value: false +bf16: + value: true +bf16_full_eval: + value: false +bos_token_id: + value: 1 +chunk_size_feed_forward: + value: 0 +data_seed: + value: 42 +dataloader_drop_last: + value: false +dataloader_in_order: + value: true +dataloader_multiprocessing_context: + value: null +dataloader_num_workers: + value: 0 +dataloader_persistent_workers: + value: false +dataloader_pin_memory: + value: true +dataloader_prefetch_factor: + value: null +ddp_backend: + value: null +ddp_broadcast_buffers: + value: null +ddp_bucket_cap_mb: + value: null +ddp_find_unused_parameters: + value: null +ddp_static_graph: + value: null +ddp_timeout: + value: 1800 +debug: + value: [] +deepspeed: + value: null +disable_tqdm: + value: false +do_eval: + value: true +do_predict: + value: false +do_train: + value: false +dtype: + value: null +enable_jit_checkpoint: + value: false +eos_token_id: + value: 2 +eval_accumulation_steps: + value: null +eval_delay: + value: 0 +eval_do_concat_batches: + value: true +eval_on_start: + value: false +eval_steps: + value: 50 +eval_strategy: + value: steps +eval_use_gather_object: + value: false +fp16: + value: false +fp16_full_eval: + value: false +fsdp: + value: null +fsdp_config: + value: null +full_determinism: + value: false +gradient_accumulation_steps: + value: 16 +gradient_checkpointing: + value: false +gradient_checkpointing_kwargs: + value: null +greater_is_better: + value: null +head_dim: + value: 32 +hidden_act: + value: silu +hidden_size: + value: 128 +hub_always_push: + value: false +hub_model_id: + value: w-ahmad/finale-mlp-tanh-9L +hub_private_repo: + value: null +hub_revision: + value: null +hub_strategy: + value: every_save +hub_token: + value: +id2label: + value: + "0": LABEL_0 + "1": LABEL_1 +ignore_data_skip: + value: false +include_for_metrics: + value: [] +include_num_input_tokens_seen: + value: "no" +initializer_range: + value: 0.02 +intermediate_size: + value: 256 +is_encoder_decoder: + value: false +label_names: + value: null +label_smoothing_factor: + value: 0 +label2id: + value: + LABEL_0: 0 + LABEL_1: 1 +learning_rate: + value: 0.0005 +length_column_name: + value: length +liger_kernel_config: + value: null +load_best_model_at_end: + value: false +local_rank: + value: -1 +log_level: + value: passive +log_level_replica: + value: warning +log_on_each_node: + value: true +logging_first_step: + value: false +logging_nan_inf_filter: + value: true +logging_steps: + value: 20 +logging_strategy: + value: steps +lr_scheduler_kwargs: + value: null +lr_scheduler_type: + value: constant +max_grad_norm: + value: 1 +max_position_embeddings: + value: 512 +max_steps: + value: 750 +metric_for_best_model: + value: null +mlp_bias: + value: false +mlp_type: + value: mlp +model/num_parameters: + value: 2001280 +model_type: + value: tiny_llama +neftune_noise_alpha: + value: null +num_attention_heads: + value: 4 +num_hidden_layers: + value: 9 +num_key_value_heads: + value: 4 +num_train_epochs: + value: 1 +optim: + value: adamw_torch_fused +optim_args: + value: null +optim_target_modules: + value: null +output_attentions: + value: false +output_dir: + value: outio/mlp-tanh-9L_run +output_hidden_states: + value: false +pad_token_id: + value: 0 +parallelism_config: + value: null +per_device_eval_batch_size: + value: 80 +per_device_train_batch_size: + value: 80 +prediction_loss_only: + value: false +pretraining_tp: + value: 1 +problem_type: + value: null +project: + value: huggingface +push_to_hub: + value: true +remove_unused_columns: + value: false +report_to: + value: + - wandb +restore_callback_states_from_checkpoint: + value: false +resume_from_checkpoint: + value: null +return_dict: + value: true +rms_norm_eps: + value: 1e-06 +rope_parameters: + value: + rope_theta: 10000 + rope_type: default +run_name: + value: LM-mlp-tanh-9L-2.0M-20260809-061950 +save_on_each_node: + value: false +save_only_model: + value: false +save_steps: + value: 100 +save_strategy: + value: steps +save_total_limit: + value: null +seed: + value: 42 +skip_memory_metrics: + value: true +tf32: + value: null +tie_word_embeddings: + value: true +tokenizer_name: + value: w-ahmad/tiny-stories-tokenizer +torch_compile: + value: false +torch_compile_backend: + value: null +torch_compile_mode: + value: null +torch_empty_cache_steps: + value: null +trackio_bucket_id: + value: null +trackio_space_id: + value: null +trackio_static_space_id: + value: null +train_sampling_strategy: + value: random +transformers_version: + value: 5.15.0.dev0 +use_cache: + value: false +use_cpu: + value: false +use_liger_kernel: + value: false +vocab_size: + value: 4096 +warmup_steps: + value: 0 +weight_decay: + value: 0.01 diff --git a/wandb/run-20260809_061951-oe9tdw54/files/output.log b/wandb/run-20260809_061951-oe9tdw54/files/output.log new file mode 100644 index 0000000000000000000000000000000000000000..a2e618562b0a43e3653b23eab060c9d42d36e595 --- /dev/null +++ b/wandb/run-20260809_061951-oe9tdw54/files/output.log @@ -0,0 +1,7 @@ +[transformers] `use_return_dict` is deprecated! Use `return_dict` instead! +[INFO] Causal mask (float with -inf) applied to all attention layers. +/mnt/data/zainulabideen/zain-exp/notebooks/Activation/exp.py:487: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1831.) + "std": tensor.std().item(), + 6%|██▌ | 48/750 [01:37<22:37, 1.93s/it] +{'loss': '120.2', 'grad_norm': '20', 'learning_rate': '0.0005', 'epoch': '0.02696', 'train/total_time_seconds': '23.21', 'train/time_per_step_avg': '1.161', 'train/epoch_time_elapsed': '43.31', 'train/estimated_remaining_minutes': '14.12', 'train/global/act/norm': '4.583e+04', 'train/global/act/mean': '0.000972', 'train/global/act/std': '0.4056', 'train/global/act/max_abs': '8.328', 'train/global/act/frac_near_dtype_limit': '0', 'train/global/act/frac_near_user_limit': '0', 'train/global/grad/norm': '7.368', 'train/global/grad/mean': '1.736e-06', 'train/global/grad/std': '0.001302', 'train/global/grad/max_abs': '0.1855', 'train/global/grad/frac_near_dtype_limit': '0', 'train/global/grad/frac_near_user_limit': '0', 'train/global/param/norm': '56.84', 'train/global/param/mean': '0.001201', 'train/global/param/std': '0.04018', 'train/global/param/max_abs': '1', 'train/global/param/frac_near_dtype_limit': '0', 'train/global/param/frac_near_user_limit': '0', 'train/layer__model_layers_1/param/norm': '17.93', 'train/layer__model_layers_1/param/mean': '0.001505', 'train/layer__model_layers_1/param/std': '0.04425', 'train/layer__model_layers_1/param/max_abs': '1', 'train/layer__model_layers_1/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_1/param/frac_near_user_limit': '0', 'train/layer__model_layers_8/param/norm': '17.92', 'train/layer__model_layers_8/param/mean': '0.001427', 'train/layer__model_layers_8/param/std': '0.04422', 'train/layer__model_layers_8/param/max_abs': '1', 'train/layer__model_layers_8/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_8/param/frac_near_user_limit': '0', 'train/layer_model_layers_8/act/norm': '1.417e+04', 'train/layer_model_layers_8/act/mean': '0.001563', 'train/layer_model_layers_8/act/std': '0.429', 'train/layer_model_layers_8/act/max_abs': '5.594', 'train/layer_model_layers_8/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_8/act/frac_near_user_limit': '0', 'train/layer_model_layers_8/grad/norm': '1.045', 'train/layer_model_layers_8/grad/mean': '-1.41e-07', 'train/layer_model_layers_8/grad/std': '0.0006452', 'train/layer_model_layers_8/grad/max_abs': '0.007568', 'train/layer_model_layers_8/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_8/grad/frac_near_user_limit': '0', 'train/layer__model_layers_7/param/norm': '17.94', 'train/layer__model_layers_7/param/mean': '0.001528', 'train/layer__model_layers_7/param/std': '0.04426', 'train/layer__model_layers_7/param/max_abs': '1', 'train/layer__model_layers_7/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_7/param/frac_near_user_limit': '0', 'train/layer__model_layers_3/param/norm': '17.94', 'train/layer__model_layers_3/param/mean': '0.001447', 'train/layer__model_layers_3/param/std': '0.04426', 'train/layer__model_layers_3/param/max_abs': '1', 'train/layer__model_layers_3/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_3/param/frac_near_user_limit': '0', 'train/layer_model_layers_3/act/norm': '1.405e+04', 'train/layer_model_layers_3/act/mean': '0.002938', 'train/layer_model_layers_3/act/std': '0.4258', 'train/layer_model_layers_3/act/max_abs': '4.844', 'train/layer_model_layers_3/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_3/act/frac_near_user_limit': '0', 'train/layer_model_layers_3/grad/norm': '1.749', 'train/layer_model_layers_3/grad/mean': '2.299e-06', 'train/layer_model_layers_3/grad/std': '0.001079', 'train/layer_model_layers_3/grad/max_abs': '0.01093', 'train/layer_model_layers_3/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_3/grad/frac_near_user_limit': '0', 'train/layer_model_layers_7/act/norm': '1.414e+04', 'train/layer_model_layers_7/act/mean': '0.0005604', 'train/layer_model_layers_7/act/std': '0.4283', 'train/layer_model_layers_7/act/max_abs': '5.594', 'train/layer_model_layers_7/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_7/act/frac_near_user_limit': '0', 'train/layer_model_layers_7/grad/norm': '1.092', 'train/layer_model_layers_7/grad/mean': '-5.479e-07', 'train/layer_model_layers_7/grad/std': '0.0006737', 'train +{'loss': '102.1', 'grad_norm': '39.5', 'learning_rate': '0.0005', 'epoch': '0.05393', 'train/total_time_seconds': '42.02', 'train/time_per_step_avg': '1.05', 'train/epoch_time_elapsed': '81.35', 'train/estimated_remaining_minutes': '12.43'} diff --git a/wandb/run-20260809_061951-oe9tdw54/files/requirements.txt b/wandb/run-20260809_061951-oe9tdw54/files/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..123b15ebf857859624f7f4332e92341c8ef13fdf --- /dev/null +++ b/wandb/run-20260809_061951-oe9tdw54/files/requirements.txt @@ -0,0 +1,149 @@ +asttokens==3.0.1 +comm==0.2.3 +debugpy==1.8.21 +decorator==5.3.1 +executing==2.2.1 +nest-asyncio==1.6.0 +parso==0.8.7 +platformdirs==4.11.0 +psutil==7.2.2 +ptyprocess==0.7.0 +pure_eval==0.2.3 +Pygments==2.20.0 +pyzmq==27.1.0 +setuptools==83.0.0 +six==1.17.0 +tornado==6.5.7 +traitlets==5.15.0 +fsspec==2026.4.0 +wcwidth==0.8.2 +ipython_pygments_lexers==1.1.1 +jedi==0.20.0 +jupyter_core==5.9.1 +matplotlib-inline==0.2.2 +pexpect==4.9.0 +prompt_toolkit==3.0.53 +python-dateutil==2.9.0.post0 +stack_data==0.6.3 +wheel==0.47.0 +jupyter_client==8.9.1 +pip==26.1.2 +ipython==9.15.0 +ipykernel==7.2.0 +threadpoolctl==3.6.0 +pyparsing==3.3.2 +typing_extensions==4.15.0 +Jinja2==3.1.6 +narwhals==2.24.0 +kiwisolver==1.5.0 +joblib==1.5.3 +fonttools==4.63.0 +cycler==0.12.1 +scipy==1.17.1 +pandas==3.0.5 +contourpy==1.3.3 +scikit-learn==1.9.0 +matplotlib==3.11.1 +urllib3==2.7.0 +tqdm==4.70.0 +idna==3.18 +charset-normalizer==3.4.9 +certifi==2026.7.22 +requests==2.34.2 +seaborn==0.13.2 +uv==0.12.0 +shellingham==1.5.4 +mpmath==1.3.0 +attrs==26.1.0 +hf-xet==1.5.2 +nvidia-nccl-cu12==2.21.5 +MarkupSafe==3.0.3 +regex==2026.7.19 +importlib_metadata==9.0.0 +httpcore==1.0.9 +annotated-doc==0.0.5 +multidict==6.7.1 +aiohttp==3.14.3 +aiosignal==1.4.0 +xxhash==3.8.1 +aiohappyeyeballs==2.7.1 +mdurl==0.1.2 +cuda-toolkit==13.0.3.0 +networkx==3.6.1 +PyYAML==6.0.3 +nvidia-cufile==1.15.1.6 +typer==0.27.0 +torchaudio==2.6.0+cu124 +rich==15.0.0 +nvidia-cufft-cu12==11.2.1.3 +h11==0.16.0 +dill==0.4.1 +cuda-pathfinder==1.6.0 +filelock==3.29.0 +nvidia-nvtx-cu12==12.4.127 +httpx==0.28.1 +anyio==4.14.2 +numpy==2.4.4 +yarl==1.24.5 +click==8.4.2 +triton==3.2.0 +frozenlist==1.8.0 +zipp==4.1.0 +propcache==0.5.2 +tokenizers==0.22.2 +markdown-it-py==4.2.0 +nvidia-cuda-runtime==13.0.96 +cuda-bindings==13.3.1 +nvidia-cuda-cupti==13.0.85 +torch==2.6.0+cu124 +multiprocess==0.70.19 +pillow==12.2.0 +transformers==5.15.0.dev0 +wandb==0.28.1 +nvidia-curand==10.4.0.35 +sympy==1.13.1 +nvidia-cusparse==12.6.3.3 +nvidia-cuda-nvrtc==13.0.88 +typing-inspection==0.4.2 +nvidia-cusolver==12.0.4.66 +nvidia-cufft==12.0.0.61 +nvidia-cudnn-cu13==9.20.0.48 +nvidia-cublas==13.1.1.3 +pyarrow==25.0.0 +evaluate==0.4.6 +diffusers==0.39.0 +pydantic==2.13.4 +annotated-types==0.8.0 +protobuf==7.35.1 +sentry-sdk==2.66.1 +einops==0.8.2 +packaging==26.2 +nvidia-nvjitlink-cu12==12.4.127 +nvidia-curand-cu12==10.3.5.147 +nvidia-cusparselt-cu12==0.6.2 +nvidia-cusparse-cu12==12.3.1.170 +nvidia-cuda-runtime-cu12==12.4.127 +torchvision==0.21.0+cu124 +nvidia-cuda-nvrtc-cu12==12.4.127 +nvidia-cuda-cupti-cu12==12.4.127 +nvidia-cusolver-cu12==11.6.1.9 +nvidia-cublas-cu12==12.4.5.8 +nvidia-cudnn-cu12==9.1.0.70 +huggingface_hub==1.26.0 +datasets==5.0.1 +safetensors==0.8.0 +accelerate==1.14.0 +pydantic_core==2.46.4 +ninja==1.13.0 +autocommand==2.2.2 +backports.tarfile==1.2.0 +importlib_metadata==8.7.1 +jaraco.text==4.0.0 +jaraco.context==6.1.0 +jaraco.functools==4.4.0 +more-itertools==10.8.0 +packaging==26.0 +platformdirs==4.4.0 +tomli==2.4.0 +wheel==0.46.3 +zipp==3.23.0 diff --git a/wandb/run-20260809_061951-oe9tdw54/files/wandb-metadata.json b/wandb/run-20260809_061951-oe9tdw54/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..e748e2d9731087e5baf8fabae75390eff88a386d --- /dev/null +++ b/wandb/run-20260809_061951-oe9tdw54/files/wandb-metadata.json @@ -0,0 +1,96 @@ +{ + "os": "Linux-5.15.0-126-generic-x86_64-with-glibc2.35", + "python": "CPython 3.11.15", + "startedAt": "2026-08-09T06:19:51.990723Z", + "args": [ + "--config", + "/mnt/data/zainulabideen/zain-exp/notebooks/Activation/configs/baseline1.yaml", + "--variants", + "mlp-tanh-9L", + "--push" + ], + "program": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py", + "codePath": "sweep.py", + "codePathLocal": "sweep.py", + "git": { + "remote": "https://github.com/deepnevro/Activation.git", + "commit": "34b8d2e8f9a0c5751333310e69fa0c1056381deb" + }, + "email": "deepnevro@gmail.com", + "root": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation", + "host": "deeplens-k3s-node1", + "executable": "/mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python", + "cpu_count": 112, + "cpu_count_logical": 224, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1560765693952", + "used": "708290355200" + } + }, + "memory": { + "total": "2164089937920" + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-39c684a5-fde6-83d7-1663-0859795881ae" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-132944c4-b689-2b5f-89a4-d730401677ab" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bc6c3e3c-9b90-09ca-c034-774961847c54" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea" + } + ], + "cudaVersion": "12.4", + "writerId": "m9tdjbfp3yvliqsqunb1cen1ueg4937y" +} \ No newline at end of file diff --git a/wandb/run-20260809_061951-oe9tdw54/files/wandb-summary.json b/wandb/run-20260809_061951-oe9tdw54/files/wandb-summary.json new file mode 100644 index 0000000000000000000000000000000000000000..ea27e7b0a7648f1d4907f1265353ea69dea2a797 --- /dev/null +++ b/wandb/run-20260809_061951-oe9tdw54/files/wandb-summary.json @@ -0,0 +1 @@ +{"train/train/tensor_act_model_layers_5/std":0.21972739454715445,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/std":0.00029015083781547014,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/max_abs":0.000637054443359375,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/max_abs":0.01092529296875,"train/train/tensor_act_model_layers_7_input_layernorm/mean":0.0102386474609375,"train/train/tensor_act_model_layers_6_self_attn_o_proj/norm":253.97356640288828,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/norm":0.004203062460746068,"train/train/tensor_act_model_norm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/max_abs":0.1025390625,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_7_input_layernorm/std":1.0000039367509814,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/norm":4.4375,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_5/act/norm":14100.803074018622,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/std":0.0009851406978738572,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/norm":0.6664715322380179,"train/train/layer__model_layers_6/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/norm":0.00228196070448423,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/max_abs":0.008056640625,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/mean":6.67572021484375e-05,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/mean":-3.0529918149113655e-08,"train/train/tensor_act_model_layers_1_mlp_up_proj/std":0.22558610679893412,"train/train/layer__model_layers_8/param/mean":0.001426565851696568,"_runtime":97,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/max_abs":0.08837890625,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/mean":5.669891834259033e-06,"train/train/tensor_act_model_layers_5_mlp_up_proj/norm":3573.309552268948,"train/train/tensor_act_model_layers_2_mlp_down_proj/std":0.082367886029098,"train/train/layer_model_layers_8/act/mean":0.0015629919675680308,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/norm":0.6598499196816698,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/std":0.0018834577105885449,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/std":0.019775390625,"train/train/tensor_act_model_layers_2_mlp_up_proj/mean":0.0017571449279785156,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/std":0.0007510421049635982,"train/train/tensor_act_model_layers_6/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_input_layernorm_weight/mean":1,"train/train/global/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/mean":0.0017676353454589844,"train/train/tensor_act_model_layers_4_self_attn_q_proj/mean":-0.009613037109375,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_5_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_v_proj/mean":0.008531570434570312,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_k_proj/max_abs":1.0234375,"train/train/tensor_act_model_layers_1_self_attn_k_proj/std":0.22747849920145083,"train/train/tensor_act_model_layers_1_self_attn/norm":180.36924706885924,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/max_abs":0.083984375,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/max_abs":0.07666015625,"train/train/layer_model_layers_2/grad/std":0.0014146170196390482,"train/train/tensor_act_model_layers_1_mlp/mean":0.0012128353118896484,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/mean":-0.00011968612670898438,"train/train/tensor_param_model_layers_2_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/max_abs":0.080078125,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/max_abs":0.006011962890625,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/std":0.0008547169237887492,"train/train/tensor_act_model_layers_6_mlp_up_proj/frac_near_user_limit":0,"train/train/layer_model_layers_6/act/std":0.42734430023274333,"train/train/tensor_act_model_layers_2_self_attn_k_proj/max_abs":1.3046875,"train/train/tensor_act_model_embed_tokens/std":0.02001953329041244,"train/train/layer__model_layers_8/param/norm":17.918659803138876,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/mean":5.2871182560920715e-06,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_4/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/mean":-1.6802921891212463e-05,"train/train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/mean":8.410215377807617e-05,"train/train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_embed_tokens_weight/mean":4.4345855712890625e-05,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_7_post_attention_layernorm/norm":9158.836975108354,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_0/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/max_abs":0.07861328125,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_7_self_attn_v_proj/max_abs":1.1953125,"train/train/tensor_act_model_layers_5_self_attn_k_proj/max_abs":1.171875,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_5/grad/mean":-1.1199119906907382e-06,"train/train/tensor_act_model_layers_5_mlp/mean":-0.00046265125274658203,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_k_proj/norm":2051.8907096354474,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/max_abs":0.01214599609375,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_2/act/norm":14024.082295312,"train/train/tensor_act_model_layers_6_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/mean":3.700843080878258e-07,"train/train/tensor_act_model_layers_5_self_attn/mean":0.0007239580154418945,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/norm":4.4375,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/norm":0.020683129569258243,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/std":0.001702762793163496,"train/train/tensor_act_model_layers_3_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_2/param/norm":17.933035158611606,"train/train/tensor_act_model_layers_8_mlp_down_proj/mean":-0.004505157470703125,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/mean":-3.460794687271118e-05,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_mlp/max_abs":0.439453125,"train/train/tensor_act_model_layers_7_self_attn_q_proj/std":0.22870093704120772,"train/train/tensor_param_model_layers_5_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_7/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/std":0.0001656018060285309,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/max_abs":0.0016326904296875,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8/frac_near_dtype_limit":0,"train/train/tensor_grad_model_norm_weight/max_abs":0.0029296875,"train/train/tensor_act_/norm":33.30498616499406,"train/train/tensor_act_model_layers_4_self_attn_o_proj/mean":0.0005696415901184081,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/mean":8.869171142578125e-05,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/max_abs":0.0010833740234375,"train/train/tensor_act_model_rotary_emb/frac_near_user_limit":0,"train/train/layer_model_layers_7/grad/std":0.0006737248985529821,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/mean":0.00010395050048828125,"train/train/tensor_act_model_layers_5_self_attn_v_proj/std":0.22674686727818974,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/norm":1.0447064614408066,"train/train/layer__model_layers_0/param/mean":0.0015731311625511895,"train/train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/mean":-8.296966552734375e-05,"train/train/layer_model_layers_5/grad/max_abs":0.00872802734375,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/std":0.0007638811556086561,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/mean":1.737847924232483e-05,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp/max_abs":0.4453125,"train/train/tensor_act_model_layers_8_mlp_up_proj/max_abs":1.3828125,"train/train/tensor_act_model_layers_2_self_attn_q_proj/std":0.22515963858329976,"train/train/tensor_act_model_layers_1_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_0_self_attn_k_proj/std":0.2281499629770971,"train/train/tensor_act_model_layers_5_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/mean":-0.0002689361572265625,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_0_input_layernorm/mean":-0.0023627281188964844,"train/train/tensor_act_model_layers_8_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_q_proj/max_abs":1.203125,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_4/act/norm":14065.020087062054,"train/train/layer_model_layers_6/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_o_proj/norm":226.46827816100497,"train/train/tensor_act_model_layers_4_self_attn_o_proj/max_abs":0.236328125,"train/train/tensor_act_model_layers_5_input_layernorm/norm":9158.807006839641,"train/loss":102.0810546875,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/std":0.22851674734790905,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_0/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/norm":0.6772388768653869,"train/train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_lm_head/max_abs":1.3203125,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/norm":0.003984350401320912,"train/train/tensor_act_model_layers_4_mlp_up_proj/norm":3555.220216731526,"train/train/estimated_remaining_minutes":12.42977086493435,"train/train/tensor_act_model_layers_4_self_attn/max_abs":0.236328125,"train/train/tensor_act_model_layers_4/max_abs":1.109375,"train/train/tensor_act_model_layers_8_self_attn_v_proj/std":0.22577102326355847,"train/train/tensor_act_model_layers_6_post_attention_layernorm/std":1.0000044143605804,"train/train/tensor_act_model_layers_2/mean":0.004010200500488282,"train/train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_up_proj/max_abs":1.21875,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/norm":4.4375,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/max_abs":0.0888671875,"train/train/tensor_act_model_layers_5_post_attention_layernorm/mean":0.00972747802734375,"train/train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn/max_abs":0.2265625,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp/std":0.08636519797159739,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/max_abs":0.087890625,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/mean":9.417533874511719e-06,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/mean":-0.00014400482177734375,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/mean":-8.344650268554688e-05,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/mean":3.3855438232421875e-05,"train/train/layer_model_layers_5/act/frac_near_user_limit":0,"train/train/layer__model_layers_1/param/std":0.04424885441992243,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_up_proj/norm":3595.762705617342,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/max_abs":0.000766754150390625,"train/train/layer_model_layers_5/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2/max_abs":0.79296875,"train/train/tensor_act_model_layers_4_mlp/mean":-0.0013844966888427734,"train/train/tensor_act_model_layers_2_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_v_proj/max_abs":1.234375,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_4/act/mean":0.001356604007574228,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/norm":0.007680140530203822,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_up_proj/norm":3560.757550565122,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_act_model_embed_tokens/norm":183.14187215122737,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/norm":0.32286321546290747,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/norm":0.017560181466391808,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/norm":0.013077298558122497,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/max_abs":0.01019287109375,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/mean":-1.3563408174377402e-08,"train/train/tensor_act_model_layers_7_self_attn_k_proj/max_abs":1.2265625,"train/train/tensor_act_model_layers_5_self_attn_v_proj/max_abs":1.2578125,"train/train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/act/max_abs":4.84375,"train/train/tensor_grad_model_norm_weight/norm":0.04232738579843535,"train/train/layer__model_layers_4/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/norm":0.047971691451455564,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/std":0.0003619688785404763,"train/train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_8/grad/max_abs":0.007568359375,"train/train/tensor_act_model_layers_1_mlp_up_proj/norm":3579.792003483908,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/mean":2.6007983251474798e-06,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/mean":0.0002117156982421875,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/mean":-1.7476268112659454e-06,"train/train/tensor_act_model_layers_3_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/norm":2065.720629302085,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/mean":-1.507229171693325e-06,"train/train/tensor_act_model_layers_6_self_attn/norm":253.97356640288828,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/mean":0.0007239580154418945,"train/train/tensor_act_model_layers_3/frac_near_user_limit":0,"train/train/layer__model_layers_0/param/norm":17.93527452440093,"train/train/tensor_act_model/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/max_abs":0.08740234375,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/max_abs":0.08447265625,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_embed_tokens_weight/std":0.02001953125,"train/train/tensor_param_model_layers_1_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/mean":-1.7128513718489558e-08,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/norm":0.01663034412528906,"train/train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/mean":1.8830178305506706e-08,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/std":0.0010066488793133873,"train/train/tensor_act_model_layers_0/std":0.0869143314307497,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/max_abs":5.6743621826171875e-05,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/max_abs":0.00095367431640625,"train/train/tensor_param_model_embed_tokens_weight/norm":14.4375,"train/train/tensor_act_model_layers_1/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/norm":0.01634202061296293,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/std":0.003980138465929412,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/max_abs":0.007049560546875,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/norm":4.4375,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_k_proj/norm":2100.7299753273983,"train/train/tensor_act_model_layers_7_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/mean":-0.0002079010009765625,"train/train/layer_model_layers_6/act/norm":14112.49991284504,"train/train/tensor_act_model_layers_0_post_attention_layernorm/max_abs":4.75,"train/train/tensor_act_model_layers_3_self_attn_v_proj/norm":2072.680207254902,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_lm_head/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer__model_layers_0/param/std":0.04425206676629325,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_v_proj/std":0.22186457135269577,"train/train/tensor_act_model_layers_4_mlp_down_proj/std":0.0845341598013693,"train/train/tensor_act_model_layers_5_self_attn_q_proj/mean":0.011692047119140625,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/mean":2.5192275643348694e-06,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/mean":7.601454854011536e-06,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/norm":0.5041465482034263,"train/train/layer__model_layers_0/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5/max_abs":1.2109375,"train/train/tensor_act_model_layers_2_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_0/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/norm":0.018183048129474702,"train/train/tensor_act_model_layers_5_post_attention_layernorm/std":1.000002252895874,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_7_self_attn_v_proj/norm":2033.2786938621066,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/max_abs":0.0859375,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/mean":-8.20159912109375e-05,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/mean":-4.3190084397792816e-08,"train/train/tensor_act_model_layers_5_mlp/std":0.08306951397613167,"train/train/tensor_param_model_layers_8_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/norm":0.0017955851347056107,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/norm":0.3361567292025176,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_6_self_attn_o_proj/std":0.027715097177038607,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/max_abs":0.00177001953125,"train/train/tensor_act_model_layers_7_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_q_proj/max_abs":1.1328125,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/std":0.02001953125,"train/train/layer_model_layers_4/grad/mean":-2.1866095570524845e-06,"train/train/tensor_act_model_layers_0_self_attn_q_proj/norm":2073.7606174205857,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/mean":8.533243089914326e-08,"train/train/tensor_act_model_rotary_emb/norm":3632.2067871093745,"train/train/layer_model_layers_8/grad/std":0.0006452044131179452,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_3_post_attention_layernorm/max_abs":4.84375,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/norm":1.3890934772277073,"train/train/tensor_act_model_layers_6_self_attn_v_proj/norm":2090.723313709615,"train/train/tensor_act_model_layers_6_self_attn/std":0.027715097177038607,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/max_abs":0.00118255615234375,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_0_mlp/std":0.08425967874819335,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/max_abs":0.0869140625,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_0/param/max_abs":1,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/std":0.0038570515437090767,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/std":4.456957275638027e-06,"train/train/tensor_act_model_layers_1_self_attn_q_proj/max_abs":1.1875,"train/train/tensor_act_model_layers_2_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2/std":0.14959821173642976,"train/train/tensor_act_model_layers_7_self_attn/mean":-0.0016374588012695312,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/max_abs":0.09033203125,"train/train/tensor_act_model_layers_3_self_attn_v_proj/std":0.22631944523624664,"train/train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_5/act/std":0.427023307556937,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/max_abs":6.341934204101562e-05,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/max_abs":0.0294189453125,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_5/param/std":0.04425301650466886,"train/train/tensor_act_model_layers_3_self_attn_q_proj/max_abs":1.1875,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/norm":0.893070010286946,"train/epoch":0.053926525109538256,"train/train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_q_proj/std":0.22436635536725583,"train/train/tensor_act_model_layers_0_self_attn_v_proj/max_abs":1.0703125,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/max_abs":0.08056640625,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/std":0.0008511400353156234,"train/train/tensor_act_model_layers_4_self_attn_o_proj/std":0.027843332063912866,"train/train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/max_abs":0.00010251998901367188,"train/train/tensor_param_model_layers_1_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_7_mlp/std":0.08227610630735949,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/std":5.138961310076266e-06,"train/train/tensor_act_model_layers_8_self_attn/norm":258.71556009674106,"train/train/tensor_act_model_layers_4_self_attn_k_proj/mean":0.006021499633789062,"train/train/tensor_act_model_layers_1_post_attention_layernorm/std":1.0000032041136646,"train/train/tensor_act_model_layers_2_mlp_up_proj/std":0.2249152545309565,"train/train/tensor_param_model_layers_3_input_layernorm_weight/std":0,"train/train/layer_model_layers_3/grad/norm":1.7488206388186616,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/std":0.22937136367933203,"train/train/tensor_act_model_layers_5_mlp_up_proj/std":0.2254035162165185,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/max_abs":0.00531005859375,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/norm":0.0026383203040014954,"train/train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model/norm":9158.858032251153,"train/train/tensor_act_model_layers_4_self_attn/norm":255.11202588150823,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_4_post_attention_layernorm/std":1.0000021569116524,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/norm":0.36997189853249035,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/mean":1.628650352358818e-07,"train/train/layer__model_layers_3/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_up_proj/norm":3566.9119113736056,"train/train/tensor_act_model_layers_1_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_rotary_emb/mean":0.333984375,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/mean":2.362998202443123e-06,"train/train/tensor_act_model_layers_6/max_abs":1.46875,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/norm":0.002163481600807979,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/std":5.003466769153688e-06,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/std":0.0010431210896587194,"train/train/tensor_act_model_layers_2_self_attn_q_proj/mean":-0.012233734130859375,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/max_abs":4.982948303222656e-05,"train/train/layer_model_layers_7/act/mean":0.0005603524354787973,"train/train/tensor_act_model_layers_2_self_attn_v_proj/std":0.228882807276572,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_4_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_norm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/max_abs":5.09375,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/std":0.019775390625,"train/train/layer_model_layers_5/grad/std":0.0007662412491179434,"train/train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/std":0.0007443170822539778,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/mean":-5.269050598144531e-05,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/std":0.0011842580330185134,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/max_abs":0.0908203125,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/max_abs":0.0052490234375,"train/train/layer_model_layers_1/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/max_abs":0.002532958984375,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp_up_proj/max_abs":1.1796875,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_1_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_3_mlp_down_proj/std":0.08636519797159739,"train/train/tensor_act_model_layers_2_mlp_up_proj/max_abs":1.2734375,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/std":5.038296527592207e-06,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/max_abs":0.00030517578125,"train/train/tensor_act_model_layers_0/mean":-0.0008854866027832031,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/max_abs":0.08642578125,"train/train/tensor_act_model_layers_6_self_attn_k_proj/std":0.22381882679304918,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/max_abs":0.005157470703125,"train/train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_input_layernorm/norm":9158.843505887427,"train/train/epoch_time_elapsed":81.34817474707961,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/std":0.0006565563366839895,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/norm":0.3149092310357127,"train/train/tensor_act_model_layers_3_self_attn/max_abs":0.2265625,"train/train/tensor_act_model_layers_5_input_layernorm/std":1.000002211309776,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/mean":-1.239473931491375e-06,"train/train/layer__model_layers_8/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_4/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/mean":7.618218660354614e-06,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_0_self_attn_o_proj/mean":0.0001543760299682617,"train/train/tensor_act_model_layers_3_input_layernorm/norm":9158.712524420385,"train/train/tensor_act_model_layers_6_self_attn_v_proj/max_abs":1.28125,"train/train/global/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/norm":0.48769862701304106,"train/train/tensor_act_model_layers_6_mlp/std":0.08557226024157712,"train/train/tensor_act_model_layers_2_input_layernorm/std":1.0000010819343121,"train/train/tensor_param_model_layers_0_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_2_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp/max_abs":0.4609375,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_user_limit":0,"train/train/global/grad/std":0.0013023133279015384,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_norm/max_abs":5.34375,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/mean":1.914799213409424e-06,"train/train/tensor_act_model_layers_6_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_embed_tokens/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_3/param/mean":0.0014466554995817996,"train/train/layer_model_layers_8/act/std":0.42900728525991605,"train/train/tensor_act_model_layers_4_self_attn_k_proj/std":0.2294319808690155,"train/train/tensor_act_model_layers_1_input_layernorm/std":1.000003035874154,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_7/param/mean":0.001527878497952418,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_7/act/norm":14142.08189676352,"train/train/tensor_act_model_layers_8_mlp_down_proj/norm":772.4732071678798,"train/train/tensor_act_model_layers_4_mlp/std":0.0845341598013693,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/norm":2.53125,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/max_abs":0.09033203125,"train/train/layer__model_layers_6/param/std":0.04422801786462898,"train/train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/mean":4.575587809085846e-06,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_2/grad/norm":2.2923954362436416,"train/train/layer_model_layers_0/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/mean":-1.55740735863219e-08,"train/train/tensor_act_model_layers_7_self_attn_q_proj/max_abs":1.125,"train/train/tensor_param_model_layers_1_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/max_abs":0.001953125,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/std":0.0002329729119698576,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_1_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/max_abs":0.00537109375,"train/train/tensor_act_model_layers_3_mlp_up_proj/std":0.22656262048581174,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/mean":-0.00010919570922851562,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8/norm":2442.3700282883,"train/train/layer__model_layers_6/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/max_abs":1.1484375,"train/train/tensor_act_model_layers_1_self_attn_v_proj/norm":2049.253818858697,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_2/act/mean":0.0027060508728027344,"train/train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp/max_abs":0.453125,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/mean":-8.720671758055687e-07,"train/train/tensor_act_model_rotary_emb/std":0.71875,"train/train/tensor_param_model_layers_2_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_down_proj/max_abs":0.5078125,"train/train/tensor_act_model_layers_6/std":0.2349867621171829,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/mean":2.4847686290740967e-05,"train/train/layer_model_layers_0/grad/std":0.0025923967714678426,"train/train/global/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_0/act/std":0.4233368714469234,"train/train/tensor_act_model_layers_4_input_layernorm/std":1.0000035008045074,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/std":3.507007765161376e-06,"train/train/tensor_act_model_layers_1_mlp_down_proj/max_abs":0.453125,"train/train/tensor_act_model_layers_4_post_attention_layernorm/max_abs":5.375,"train/train/tensor_act_model_layers_0_self_attn/max_abs":0.2578125,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/std":5.342103183918348e-06,"train/train/tensor_act_model_layers_5/norm":2012.5868662278626,"train/train/tensor_act_model_layers_1_input_layernorm/max_abs":5.28125,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_5_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/norm":753.447170392316,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/std":7.895382470794292e-06,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/std":0.0009525365850671751,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/std":0.00030603420712215727,"train/train/tensor_act_model_layers_5_post_attention_layernorm/norm":9158.804016119773,"train/train/tensor_param_model_layers_8_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/norm":0.7572818746725153,"train/train/tensor_act_model_layers_8_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_0/act/norm":13963.784289654679,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/norm":0.008047597318007088,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/std":1.1136478952345827e-05,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/norm":4.40625,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/norm":2103.186552565286,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_o_proj/mean":0.0012559890747070312,"train/train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/mean":-6.67572021484375e-05,"train/train/time_per_step_avg":1.050403171684593,"train/train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp/frac_near_user_limit":0,"train/train/tensor_act_lm_head/std":0.22656261470519184,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/max_abs":0.08349609375,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_3_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/max_abs":0.00872802734375,"train/train/tensor_act_model/std":1.000004099419111,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_/mean":8.326246440410616,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_0/grad/mean":5.849929514280203e-09,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/norm":1.359781840161146,"train/train/tensor_act_model_layers_3_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_0/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/mean":-0.0006031990051269532,"train/train/layer_model_layers_4/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/max_abs":0.01226806640625,"train/train/tensor_act_model_layers_8_mlp/std":0.08425981926404569,"train/train/tensor_act_model_layers_8_self_attn_v_proj/norm":2069.6288876369786,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_4/std":0.19793796607985248,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp_down_proj/norm":783.6076718471597,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/norm":4.40625,"train/train/tensor_act_model_layers_1/norm":1124.2650574824988,"train/train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp/mean":0.0012459754943847656,"train/train/tensor_act_model_layers_8_self_attn_k_proj/max_abs":1.3046875,"train/train/tensor_act_model_layers_1_self_attn_k_proj/max_abs":1.1640625,"train/train/tensor_param_model_layers_4_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_2_mlp/std":0.082367886029098,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/mean":0.00013065338134765625,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/mean":7.382477633655071e-07,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/mean":0.000164031982421875,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_7_self_attn_v_proj/mean":-0.0056133270263671875,"train/train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/mean":1.0567717254161835e-05,"train/train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/max_abs":0.00145721435546875,"train/train/tensor_act_model_layers_1_self_attn_o_proj/norm":180.36924706885924,"train/train/layer__model_layers_3/param/max_abs":1,"train/train/tensor_param_model_layers_5_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/norm":0.004042434530392268,"train/train/layer_model_layers_3/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/std":0.0007226019432821048,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit":0,"_wandb":{"runtime":97},"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_8/param/std":0.04421874775973169,"train/train/tensor_act_model_layers_2_self_attn_k_proj/norm":2092.7739870028217,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/max_abs":0.08349609375,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_q_proj/norm":2065.6788890504995,"train/train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_1/param/max_abs":1,"train/train/tensor_act_model_layers_3_self_attn_v_proj/max_abs":1.1796875,"train/train/tensor_grad_model_norm_weight/mean":0.00026679039001464844,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/std":4.221860208729979e-06,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/mean":-3.399909473955631e-07,"train/train/tensor_act_model_layers_6_self_attn_v_proj/mean":0.014347076416015625,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/mean":-3.5762786865234375e-05,"train/train/tensor_act_model_layers_0/norm":795.7147017678604,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/global/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/max_abs":1.0703125,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/norm":0.4385958978582016,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_norm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/std":0.0006633665069236685,"train/train/layer_model_layers_3/grad/max_abs":0.01092529296875,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/max_abs":0.0023345947265625,"train/train/layer__model_layers_7/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_v_proj/max_abs":1.2109375,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/norm":0.007552827001683844,"train/train/layer_model_layers_0/act/max_abs":4.75,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/max_abs":0.006103515625,"train/train/layer_model_layers_1/grad/max_abs":0.0184326171875,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_6_mlp/max_abs":0.5078125,"train/train/layer__model_layers_1/param/mean":0.00150473590202153,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/norm":1.0505652223336508,"train/train/tensor_act_model_layers_1/max_abs":0.609375,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp_down_proj/std":0.08551056605861647,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_post_attention_layernorm/max_abs":4.875,"train/train/tensor_act_model_layers_6_input_layernorm/norm":9158.826354986366,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/std":3.609412057366975e-06,"train/train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_v_proj/max_abs":1.1796875,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/mean":2.0489096641540525e-07,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/norm":0.013838451457588039,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_post_attention_layernorm/std":1.000005643071284,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/max_abs":0.416015625,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/norm":0.01809298647783285,"train/train/global/grad/mean":1.7360134756876538e-06,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/std":0.00038804992747932545,"train/train/tensor_act_model_layers_0_self_attn_k_proj/mean":8.234567940235139e-05,"train/train/tensor_act_model_layers_5_mlp_up_proj/max_abs":1.171875,"train/train/tensor_act_model_layers_7_post_attention_layernorm/std":1.0000039266957168,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_1/grad/mean":2.754121628704709e-06,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_up_proj/max_abs":1.2578125,"train/train/tensor_act_model_layers_8_self_attn_o_proj/norm":258.71556009674106,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/max_abs":6.723403930664062e-05,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/std":0.0025358019538881085,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/mean":-0.00013446807861328125,"train/train/tensor_act_model_layers_8_self_attn/std":0.028224869081951577,"train/train/tensor_act_model_layers_1_mlp/std":0.08551056605861647,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/norm":0.3936700770084772,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_1/grad/std":0.0016707387393106302,"train/train/global/param/max_abs":1,"train/train/tensor_param_model_layers_2_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_6/grad/std":0.0007126712014498404,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_5/grad/norm":1.241541045998697,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_6/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_norm_weight/std":0,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_2_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3/norm":1608.0517896816252,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/std":0.008522998259386573,"train/train/layer_model_layers_3/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/std":0.0004031774962814704,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_k_proj/mean":0.008234024047851562,"train/train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_input_layernorm/mean":0.01171112060546875,"train/train/layer__model_layers_1/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_k_proj/max_abs":1.1328125,"train/train/tensor_act_model_layers_7_post_attention_layernorm/max_abs":5.4375,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/max_abs":0.00139617919921875,"train/train/tensor_act_model_layers_0_mlp/norm":771.7085156076147,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/max_abs":0.07958984375,"train/train/tensor_act_model_layers_8_self_attn/mean":0.0008984804153442383,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_5/mean":0.0015457868576049805,"train/train/layer_model_layers_4/act/std":0.4259628077063781,"train/train/layer_model_layers_6/grad/norm":1.154570131496639,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/norm":0.002265625107904958,"train/train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/mean":-3.879540599882603e-08,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/mean":-0.0002593994140625,"train/train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_up_proj/std":0.22656270507346202,"train/train/tensor_param_model_embed_tokens_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_6/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_1_input_layernorm/norm":9158.302429209018,"train/train/tensor_act_model_layers_4_input_layernorm/mean":0.011629104614257812,"train/train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/max_abs":0.0830078125,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/norm":0.0038438257099948896,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/mean":0.00015544891357421875,"train/train/tensor_act_model_layers_1_self_attn_o_proj/std":0.01964113700617179,"train/train/tensor_param_model_layers_7_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/max_abs":0.00011539459228515625,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/std":0.0006305925327723769,"train/train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_norm/norm":9158.858032251153,"train/train/tensor_act_model_layers_2_self_attn_q_proj/max_abs":1.3828125,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/mean":-0.00010156631469726562,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/max_abs":0.0001087188720703125,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn/mean":0.0011792182922363281,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_input_layernorm_weight/std":0,"train/train/layer_model_layers_3/act/norm":14054.438547142747,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/max_abs":0.076171875,"train/train/tensor_act_model_layers_5_self_attn_q_proj/max_abs":1.203125,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/norm":0.6356109155465426,"train/train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_3_self_attn_v_proj/mean":-0.006090164184570312,"train/train/tensor_act_model_layers_7_post_attention_layernorm/mean":0.0032741427421569824,"train/train/tensor_act_model_layers_2_self_attn/std":0.024912951683477253,"train/train/tensor_act_model_layers_5_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp/mean":0.0017676353454589844,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp/mean":0.0002854466438293457,"train/train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/mean":-4.696846008300781e-05,"train/train/tensor_act_model_layers_8_mlp_up_proj/mean":0.000231027603149414,"train/train/global/param/mean":0.001200859135700027,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_0_self_attn_k_proj/norm":2089.1339244416013,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/max_abs":0.007293701171875,"train/train/layer__model_layers_4/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/global/act/norm":45825.67915864439,"train/train/tensor_act_model_layers_4_self_attn/mean":0.0005696415901184081,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/mean":8.678436279296875e-05,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/std":0.001651296646585802,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/std":0.0011766864355587643,"train/train/tensor_act_model_layers_7_self_attn_o_proj/norm":249.8859830154156,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/max_abs":0.08056640625,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/mean":-4.3795444071292877e-07,"train/train/tensor_act_model_layers_5_self_attn_k_proj/mean":0.0010291337966918945,"train/train/global/act/mean":0.0009720008240165799,"train/train/tensor_act_model_layers_0/max_abs":0.50390625,"train/train/tensor_act_lm_head/norm":11726.645165878956,"train/train/tensor_grad_model_embed_tokens_weight/std":0.0012787561777355767,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/norm":0.002734198734869189,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/max_abs":5.078315734863281e-05,"train/train/tensor_act_model_layers_1_self_attn_v_proj/mean":-0.0009049177169799805,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp_up_proj/norm":3593.418878817968,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/mean":7.370021194219589e-06,"train/train/tensor_act_model_layers_0_post_attention_layernorm/std":1.000002007208682,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp_down_proj/mean":0.0012128353118896484,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn/std":0.027843332063912866,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/std":0.0008566329475932972,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/mean":-6.29425048828125e-05,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/mean":-5.511566996574402e-06,"train/train/tensor_act_model_layers_1/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_up_proj/max_abs":1.234375,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_0_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_1_post_attention_layernorm/max_abs":5.28125,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/max_abs":0.0184326171875,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/norm":0.03457925756213994,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/max_abs":0.01446533203125,"_step":1,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/mean":-2.3712345864623785e-08,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/max_abs":7.390975952148438e-05,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/std":0.0018290869677934636,"train/train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/std":1.000000091723332,"train/train/tensor_param_model_layers_6_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_down_proj/norm":753.8613409709429,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/std":0.001859627934527953,"train/train/layer_model_layers_0/grad/norm":4.200656564761473,"train/train/tensor_act_model_layers_0_mlp/max_abs":0.490234375,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_0_post_attention_layernorm/mean":0.004792213439941406,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/mean":-3.2745301723480225e-06,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/max_abs":0.080078125,"train/train/layer__model_layers_8/param/max_abs":1,"train/train/tensor_act_model_layers_6_mlp_down_proj/mean":0.0002854466438293457,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_6/act/mean":0.0014113554587730994,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/mean":6.875779945403336e-08,"train/train/layer_model_layers_4/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/norm":773.9119460845479,"train/train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_input_layernorm/std":1.0000030592426397,"train/train/tensor_act_model_layers_0_self_attn_q_proj/std":0.2265625557654413,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/mean":-2.0142178982496257e-06,"train/train/tensor_param_model_norm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_embed_tokens/mean":-3.91155481338501e-05,"train/train/layer_model_layers_3/grad/std":0.0010787107165929604,"train/train/tensor_act_model_layers_2_self_attn_k_proj/std":0.22839403667703648,"train/train/tensor_act_model_layers_8_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/std":0.22442723998789932,"train/train/layer_model_layers_7/grad/max_abs":0.00836181640625,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_7/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/std":0.000161797306189962,"train/train/tensor_act_model_layers_1_self_attn/max_abs":0.2412109375,"train/train/tensor_act_model_layers_5_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_8_self_attn_k_proj/norm":2066.244783511504,"train/train/tensor_act_model_layers_1_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/std":0.02001953125,"train/train/layer__model_layers_5/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/mean":-0.00020503997802734375,"train/train/layer_model_layers_8/grad/mean":-1.409528021484195e-07,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/norm":0.01052831013690247,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_8/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/norm":4.4375,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/mean":8.304341463372111e-07,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/max_abs":0.001220703125,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/std":0.0002797351357872544,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/mean":1.7954735085368156e-06,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/mean":3.390759229660034e-05,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/mean":-9.679794311523438e-05,"train/train/tensor_act_model_layers_4_input_layernorm/norm":9158.770935065297,"train/train/tensor_act_model_layers_3/mean":0.002099037170410156,"train/train/tensor_act_model_layers_1_input_layernorm/mean":-0.0088043212890625,"train/train/tensor_act_model_layers_6_mlp_up_proj/std":0.22460983234978,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_k_proj/mean":0.0003761379048228264,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/mean":3.814697265625e-05,"train/train/layer_model_layers_5/grad/frac_near_user_limit":0,"train/train/total_time_seconds":42.01612686738372,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_4_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_3/param/std":0.04426201158065172,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/mean":-3.9261067286133766e-08,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/mean":-0.0002498626708984375,"train/train/tensor_act_model_layers_5_input_layernorm/max_abs":5.1875,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/mean":9.191036224365232e-05,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/norm":0.8457144016280518,"train/train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/norm":4.4375,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/mean":1.8924474716186523e-06,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/std":0.0201416015625,"train/train/layer__model_layers_4/param/norm":17.9352881367118,"train/train/layer_model_layers_2/act/std":0.4247193027941706,"train/train/tensor_act_model_layers_8_mlp/mean":-0.004505157470703125,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/mean":-0.00015735626220703125,"train/train/layer_model_layers_3/act/mean":0.002938002347946167,"train/train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_embed_tokens_weight/max_abs":0.185546875,"train/train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/norm":1.974809313876121,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/max_abs":0.0791015625,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/max_abs":0.083984375,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_3/act/std":0.42579208331501933,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/mean":-9.72747802734375e-05,"train/train/layer_model_layers_3/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_7/max_abs":1.53125,"train/train/tensor_act_model_layers_6_post_attention_layernorm/norm":9158.82855225413,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/std":0.000688121658388342,"train/train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn/std":0.008522998259386573,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/mean":-1.2278556823730469e-05,"train/train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/norm":1.6502842976104004,"train/train/tensor_act_model_layers_2_input_layernorm/max_abs":4.875,"train/train/layer_model_layers_1/grad/frac_near_user_limit":0,"train/train/layer_model_layers_8/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/max_abs":6.818771362304688e-05,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/std":0.0008175833184701628,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_7_input_layernorm/max_abs":5.59375,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/max_abs":3.981590270996094e-05,"train/train/tensor_act_model_layers_8/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/norm":0.007290304662027054,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/mean":-3.427267074584961e-06,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/max_abs":0.078125,"train/train/tensor_act_model_layers_2_self_attn/norm":228.47226970029095,"train/train/tensor_act_model_layers_2_input_layernorm/norm":9158.614257822699,"train/train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_0/grad/max_abs":0.037353515625,"train/train/tensor_act_model_layers_8/mean":-0.001019805669784546,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/max_abs":0.009765625,"train/train/layer__model_layers_7/param/norm":17.939752417379538,"train/train/tensor_act_model_layers_6/norm":2152.4655637454157,"train/train/tensor_act_model_layers_1_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/max_abs":0.007171630859375,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/std":0.0019982397980901007,"train/train/tensor_act_model_layers_3/std":0.17572100292915102,"train/train/tensor_act_model_layers_7_mlp_down_proj/max_abs":0.4453125,"train/train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/mean":2.291053533554077e-05,"train/train/tensor_act_model_layers_4_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_down_proj/max_abs":0.443359375,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_3_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_post_attention_layernorm/norm":9158.859252948552,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/max_abs":0.08544921875,"train/train/tensor_act_model_layers_0_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/mean":-6.477679562522098e-06,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/std":0.0006155200305108026,"train/train/layer__model_layers_7/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/max_abs":5.15625,"train/train/tensor_act_model_rotary_emb/max_abs":1,"train/train/tensor_act_model_layers_2_self_attn_o_proj/max_abs":0.2265625,"train/train/layer__model_layers_3/param/frac_near_user_limit":0,"train/train/layer_model_layers_1/act/std":0.4237531971235928,"train/train/tensor_act_model_layers_8_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/std":0.0008873731221887825,"train/train/tensor_act_model_layers_8_self_attn_k_proj/mean":-0.0036034584045410156,"train/train/tensor_act_model_layers_5_self_attn_o_proj/max_abs":0.2197265625,"train/train/tensor_act_/max_abs":8.328397750854492,"train/train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_7/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/mean":6.961822509765625e-05,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/max_abs":0.07958984375,"train/train/tensor_act_model_layers_0_self_attn_o_proj/norm":78.02399889508851,"train/train/tensor_act_model_layers_1_self_attn_q_proj/std":0.22210801187994125,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/max_abs":0.0869140625,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/mean":-6.891787052154542e-05,"train/train/tensor_act_model_layers_8_mlp/max_abs":0.5,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1/std":0.12271151907771602,"train/train/tensor_act_model_layers_1_self_attn_q_proj/mean":-0.00579071044921875,"train/train/tensor_act_model_layers_8_input_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_5/param/norm":17.927609152418373,"train/train/layer_model_layers_0/act/mean":-0.00017583585129334376,"train/train/tensor_grad_model_embed_tokens_weight/mean":5.578622221946716e-06,"train/train/tensor_act_model_layers_8_mlp/norm":772.4732071678798,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_2/grad/frac_near_dtype_limit":0,"train/train/layer__model_layers_7/param/max_abs":1,"train/train/tensor_act_model_layers_8_mlp_up_proj/norm":3582.6683128871264,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/max_abs":0.00182342529296875,"train/train/tensor_act_model_layers_7/std":0.2500623991634354,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/norm":0.029950915753115565,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/std":0.019775390625,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/mean":-7.83504219725728e-07,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/max_abs":0.0174560546875,"train/train/layer_model_layers_8/grad/norm":1.0449912347693717,"train/train/tensor_act_model_layers_0_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/norm":0.03126985693910291,"train/train/tensor_act_/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/max_abs":4.172325134277344e-05,"train/train/tensor_act_model_layers_7_self_attn/max_abs":0.23046875,"train/train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/max_abs":0.007568359375,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/max_abs":0.01019287109375,"train/train/global/grad/norm":7.367828232547335,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_param_model_embed_tokens_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_embed_tokens_weight/norm":3.6993218702628035,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/std":0.002400357431670027,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/max_abs":0.006256103515625,"train/train/layer_model_layers_6/grad/mean":-6.731039354637231e-07,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/mean":-4.434026777744293e-06,"train/train/layer_model_layers_7/act/max_abs":5.59375,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/mean":7.2177499532699585e-06,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/norm":2.5625,"train/train/layer_model_layers_1/act/mean":-5.987723572896077e-05,"train/train/tensor_param_model_layers_4_input_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_6/param/mean":0.0016320834107778374,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/std":0.0004537914635213569,"train/train/tensor_act_lm_head/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/std":0.00016914082204828,"train/train/tensor_act_model_layers_0_mlp_up_proj/std":0.2246094857487788,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/std":0.0015663875075158113,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/norm":0.35389613122347957,"train/train/tensor_act_model_layers_7_self_attn_o_proj/max_abs":0.23046875,"train/train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_q_proj/norm":2094.071764344879,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_3_input_layernorm/std":1.0000036635543632,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/norm":0.005701863618059202,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/norm":0.0025617701663956053,"train/train/tensor_act_model_layers_8_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_7/act/std":0.42826948381510205,"train/train/tensor_act_model_layers_8_self_attn_k_proj/std":0.22552669692388125,"train/train/tensor_act_model_norm/std":1.000004099419111,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/mean":3.4123659133911133e-06,"train/train/layer_model_layers_2/act/max_abs":4.875,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_post_attention_layernorm/mean":0.01483917236328125,"train/train/tensor_act_model_layers_4_input_layernorm/max_abs":5.1875,"train/train/tensor_act_model_layers_0_self_attn_v_proj/norm":2107.624164923062,"train/train/layer__model_layers_2/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp/frac_near_user_limit":0,"train/train/layer__model_layers_2/param/max_abs":1,"train/train/tensor_act_model_layers_5_self_attn_k_proj/std":0.22399984632619202,"train/train/tensor_act_model_layers_8_input_layernorm/mean":0.010530471801757814,"train/train/layer__model_layers_2/param/std":0.044265280721252936,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/norm":0.0018480180721657947,"train/train/tensor_act_model_layers_1_post_attention_layernorm/mean":0.004477977752685548,"train/train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/mean":-2.535292878746986e-06,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/max_abs":0.08984375,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_7/norm":2291.0751662508405,"train/train/tensor_act_model_layers_6/mean":0.0024580955505371094,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_3_mlp/mean":-0.0032863616943359375,"train/train/tensor_act_model_layers_8/max_abs":1.5546875,"train/train/tensor_act_model_layers_3_input_layernorm/mean":0.02681732177734375,"train/train/tensor_act_model_layers_8_post_attention_layernorm/max_abs":5.53125,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_4_self_attn_v_proj/mean":-0.00487518310546875,"train/train/tensor_act_model_layers_8_self_attn_q_proj/mean":-0.0016758441925048828,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_q_proj/norm":2023.8789405812654,"train/train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/std":0.001533029250081645,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_1/act/frac_near_user_limit":0,"train/learning_rate":0.0005,"train/train/tensor_act_model_layers_2_self_attn_v_proj/norm":2097.41569083777,"train/train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_5/param/max_abs":1,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_7/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/std":0.23046878774569188,"train/train/tensor_act_model_layers_5_self_attn/std":0.028286906480645278,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/mean":-1.8071383237838745e-05,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/mean":0.00017547607421875,"train/train/global/param/std":0.04017675470731757,"train/train/tensor_act_model_layers_7_self_attn/norm":249.8859830154156,"train/train/layer_model_layers_5/act/mean":0.0020468876912043644,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/mean":2.3213215172290802e-07,"train/train/tensor_act_model_layers_2/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/norm":0.002579598592699658,"train/train/tensor_act_model_layers_5_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_2/grad/mean":2.7492822088154436e-06,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_4_self_attn_v_proj/max_abs":1.1640625,"train/train/tensor_act_model_layers_6_self_attn/mean":0.0006264448165893555,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/max_abs":0.080078125,"train/train/tensor_act_model_layers_8_self_attn_o_proj/mean":0.0008984804153442383,"train/train/layer__model_layers_5/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/max_abs":0.0045166015625,"train/train/tensor_act_model_layers_7_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_rotary_emb/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/norm":0.7875399867002663,"train/train/layer__model_layers_4/param/mean":0.001547672075340045,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/mean":-0.00018215179443359375,"train/train/tensor_act_model_layers_3_mlp_down_proj/norm":791.0893592431163,"train/train/tensor_act_model_layers_7_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/norm":2092.7443792291797,"train/train/tensor_act_model_layers_4_mlp/norm":773.9119460845479,"train/train/tensor_act_model_layers_6_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/std":0.00033254078012595953,"train/train/tensor_param_model_norm_weight/norm":11.3125,"train/train/tensor_act_model_layers_3_post_attention_layernorm/mean":0.03549957275390625,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/mean":-3.7022982724010944e-07,"train/train/tensor_act_model_layers_2/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/mean":0.0001678466796875,"train/train/layer_model_layers_7/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn/std":0.02468457451209307,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/max_abs":0.01556396484375,"train/train/tensor_act_model_layers_3_mlp_up_proj/max_abs":1.28125,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1/mean":0.0015063285827636719,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/mean":-7.304549217224121e-05,"train/train/tensor_param_model_layers_7_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_7_mlp/norm":753.447170392316,"train/train/tensor_act_model_layers_1_self_attn_v_proj/std":0.22363317047832146,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_8/act/max_abs":5.59375,"train/train/tensor_act_model_layers_1_self_attn_o_proj/mean":0.0011792182922363281,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/norm":1.622936405687018,"train/train/tensor_act_/std":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/std":0.0006912039625101131,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp_down_proj/max_abs":0.5,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/norm":0.01918604483196558,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/mean":2.7741862140828744e-09,"train/train/tensor_act_model_layers_4/mean":0.001283407211303711,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_1/act/max_abs":5.28125,"train/train/tensor_act_model_layers_4_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/max_abs":0.091796875,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_3_self_attn/frac_near_user_limit":0,"_timestamp":1.7862564739413743e+09,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/norm":0.34347939677410166,"train/train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/max_abs":0.07958984375,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/mean":-0.00020503997802734375,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/max_abs":0.0031890869140625,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/mean":-6.673508323729038e-08,"train/train/tensor_act_model_layers_7_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/std":0.0201416015625,"train/train/layer_model_layers_1/act/norm":13992.899675692339,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/max_abs":0.000736236572265625,"train/train/tensor_act_model_layers_2_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/mean":-0.0002288818359375,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/norm":0.7079498898720334,"train/train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_embed_tokens_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/norm":2078.0090258486925,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_input_layernorm_weight/std":0,"train/train/layer_model_layers_1/act/frac_near_dtype_limit":0,"train/global_step":40,"train/train/tensor_act_model_layers_4_mlp_up_proj/mean":-6.467103958129883e-06,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/std":8.205381582890514e-06,"train/train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/norm":0.004927389985235503,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_2/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_q_proj/mean":-0.0017673373222351074,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_3_self_attn_o_proj/mean":0.0013756752014160156,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/max_abs":0.005828857421875,"train/train/tensor_act_model_layers_0_mlp_down_proj/max_abs":0.490234375,"train/train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/std":0.0013227328199398023,"train/train/layer__model_layers_7/param/std":0.044264819558247244,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/mean":-3.933906555175781e-05,"train/train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/max_abs":0.00194549560546875,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/max_abs":7.534027099609375e-05,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_mlp_down_proj/std":0.08306951397613167,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/mean":-4.2983447201550007e-07,"train/train/layer_model_layers_2/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/max_abs":4.375,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/norm":4.4375,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_q_proj/mean":0.008485794067382814,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn/mean":0.0012559890747070312,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_6/param/max_abs":1,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_8_self_attn_q_proj/norm":2054.917372023513,"train/train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/mean":4.363059997558594e-05,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/max_abs":0.08056640625,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_8/act/norm":14165.892651482818,"train/train/tensor_act_model_layers_1_self_attn_q_proj/norm":2035.2156649415904,"train/train/tensor_act_model_layers_5_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_6_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/std":0,"train/train/layer_model_layers_8/act/frac_near_user_limit":0,"train/train/tensor_param_model_norm_weight/mean":1,"train/train/tensor_act_model_layers_4_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_3_self_attn/mean":0.0013756752014160156,"train/train/tensor_act_model_layers_0_mlp_up_proj/max_abs":1.1796875,"train/train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/max_abs":0.078125,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/std":9.623830616616655e-06,"train/train/tensor_param_model_layers_4_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_1_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/max_abs":0.00836181640625,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/mean":2.8848648071289062e-05,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model/max_abs":5.34375,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_4_self_attn_o_proj/norm":255.11202588150823,"train/train/layer_model_layers_6/grad/max_abs":0.00775146484375,"train/train/tensor_act_model_layers_5_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_3_post_attention_layernorm/norm":9158.71673584305,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/std":1.0000025588923567,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_up_proj/norm":3570.1849654717043,"train/train/tensor_grad_model_embed_tokens_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp_up_proj/std":0.22589204684967681,"train/train/tensor_act_model_layers_8_self_attn_o_proj/max_abs":0.25,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/norm":11.3125,"train/grad_norm":39.5,"train/train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/std":3.5337946708383607e-05,"train/train/tensor_act_model_layers_1_self_attn_o_proj/max_abs":0.2412109375,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/norm":0.7546016667998484,"train/train/tensor_act_model_layers_8/std":0.2667264865294788,"train/train/layer__model_layers_4/param/std":0.04425290086076607,"train/train/tensor_act_model_layers_4_mlp_down_proj/mean":-0.0013844966888427734,"train/train/tensor_act_model_layers_0_self_attn/mean":0.0001543760299682617,"train/train/tensor_act_model_layers_6_mlp_down_proj/norm":783.6505542822516,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_down_proj/mean":0.0012459754943847656,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/mean":9.721261449158192e-07,"train/train/tensor_act_model_layers_5_mlp/norm":761.6220634845612,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/norm":2.53125,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/max_abs":5.459785461425781e-05,"train/train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_q_proj/std":0.22082680571111493,"train/train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/max_abs":0.0830078125,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/max_abs":0.4609375,"train/train/tensor_act_model_layers_5_mlp_up_proj/mean":-0.001348257064819336,"train/train/layer__model_layers_5/param/mean":0.001554492111325078,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp/norm":783.6505542822516,"train/train/tensor_act_model_layers_0_mlp_down_proj/mean":-0.001001596450805664,"train/train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/max_abs":0.09130859375,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_1_self_attn/std":0.01964113700617179,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_5_mlp_down_proj/mean":-0.00046265125274658203,"train/train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_1/param/norm":17.933035158611606,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/norm":0.8718143783878365,"train/train/global/param/norm":56.836357923702764,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/norm":2.2479543923743894,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/std":4.425062929217393e-06,"train/train/tensor_act_model_layers_7_self_attn_o_proj/mean":-0.0016374588012695312,"train/train/layer_model_layers_2/grad/max_abs":0.01556396484375,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/max_abs":6.4849853515625e-05,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/mean":5.745887756347656e-05,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/norm":2.037830199650457,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/max_abs":0.00640869140625,"train/train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_post_attention_layernorm/norm":9158.334106452012,"train/train/tensor_act_model_layers_4_post_attention_layernorm/norm":9158.769226082162,"train/train/tensor_act_model_layers_3_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/std":0.028286906480645278,"train/train/tensor_act_model_layers_8_post_attention_layernorm/std":1.000003191608266,"train/train/tensor_act_model_layers_6_self_attn/max_abs":0.2177734375,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_o_proj/max_abs":0.2265625,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/max_abs":0.0224609375,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/max_abs":0.037353515625,"train/train/tensor_act_model_layers_8_self_attn_o_proj/std":0.028224869081951577,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp/norm":753.8613409709429,"train/train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_o_proj/mean":0.0006264448165893555,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_param_model_norm_weight/max_abs":1,"train/train/tensor_act_model_layers_4_self_attn_q_proj/std":0.22534347173992078,"train/train/tensor_act_model_layers_4_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/mean":-2.2135674953460693e-05,"train/train/tensor_act_model_layers_3_self_attn_q_proj/mean":0.0022826194763183594,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/norm":2.53125,"train/train/layer__model_layers_6/param/norm":17.92636985918022,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/norm":0.012680099615755315,"train/train/tensor_param_model_layers_7_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_6_mlp_down_proj/std":0.08557226024157712,"train/train/tensor_act_model_layers_3/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp/mean":-0.001001596450805664,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/norm":0.9250547238220178,"train/train/tensor_act_model_embed_tokens/max_abs":0.09619140625,"train/train/tensor_act_model_layers_3_mlp/norm":791.0893592431163,"train/train/tensor_act_model_layers_7_mlp_up_proj/mean":-0.0003441423177719116,"train/train/layer__model_layers_2/param/mean":0.0015104921671232083,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/mean":-1.0107214620802552e-08,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_k_proj/norm":2070.8650656116906,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/mean":-6.165355443954468e-06,"train/train/layer_model_layers_8/grad/frac_near_dtype_limit":0,"train/train/layer__model_layers_1/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/max_abs":0.01275634765625,"train/train/layer_model_layers_8/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn/std":0.02723244874144052,"train/train/tensor_act_model_layers_6_post_attention_layernorm/mean":0.010058403015136719,"train/train/layer_model_layers_7/grad/norm":1.0915601981948118,"train/train/layer_model_layers_4/grad/norm":1.4648445572552842,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/norm":0.7253473436367561,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/norm":4.4375,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/std":0.000751989631097554,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/mean":-2.0614825189113617e-06,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/max_abs":0.00885009765625,"train/train/tensor_act_model_layers_2_self_attn_v_proj/mean":-0.008272171020507812,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/mean":-8.20159912109375e-05,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/mean":-3.0443072319030762e-05,"train/train/tensor_act_model_layers_0_mlp_down_proj/std":0.08425967874819335,"train/train/layer_model_layers_2/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/max_abs":0.08251953125,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/mean":-1.704692840576172e-05,"train/train/tensor_act_model_layers_6_mlp_up_proj/mean":-0.004803657531738281,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/mean":1.7523765563964844e-05,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/std":0.0012414277473263577,"train/train/tensor_param_model_embed_tokens_weight/max_abs":0.0966796875,"train/train/tensor_act_model_layers_6_self_attn_k_proj/norm":2052.806148941093,"train/train/layer_model_layers_6/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/mean":-7.683411240577698e-09,"train/train/tensor_act_model_layers_5_post_attention_layernorm/max_abs":5.21875,"train/train/tensor_act_model_layers_5_self_attn/norm":259.09118251663875,"train/train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4/norm":1812.7927373217005,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/max_abs":3.528594970703125e-05,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/std":5.15297703171949e-06,"train/train/tensor_act_model_embed_tokens/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_q_proj/max_abs":1.1875,"train/train/tensor_act_model_norm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/global/act/max_abs":8.328397750854492,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/max_abs":0.08056640625,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/std":0.0004257842314802969,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/max_abs":0.00074005126953125,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/std":0.0007688871086946725,"train/train/layer__model_layers_4/param/max_abs":1,"train/train/tensor_act_model_layers_5_self_attn/max_abs":0.2197265625,"train/train/layer_model_layers_1/grad/norm":2.708397429771687,"train/train/layer_model_layers_4/act/max_abs":5.375,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/mean":2.8330832719802856e-06,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/max_abs":0.00823974609375,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_post_attention_layernorm/mean":0.01407623291015625,"train/train/layer_model_layers_4/grad/std":0.0009039487622426939,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/mean":-0.0032863616943359375,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/norm":4.4375,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/norm":4.4375,"train/train/tensor_act_model_layers_7_mlp_down_proj/std":0.08227610630735949,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/mean":-4.034372977912426e-07,"train/train/tensor_act_model_layers_4_mlp_up_proj/std":0.22412190403975368,"train/train/layer_model_layers_5/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_input_layernorm/norm":9158.8432006901,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/max_abs":0.01165771484375,"train/train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/mean":-0.00019931793212890625,"train/train/tensor_act_model_layers_8_self_attn/max_abs":0.25,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/max_abs":0.0791015625,"train/train/tensor_act_model_layers_2_mlp/max_abs":0.443359375,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/mean":-0.0001392364501953125,"train/train/tensor_act_model_layers_3_self_attn_k_proj/mean":-0.00013938546180725098,"train/train/tensor_act_model_layers_7/mean":0.0025863647460937496,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/std":0.0007984289616214263,"train/train/tensor_act_model_layers_6_self_attn_v_proj/std":0.22772402108894638,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/max_abs":0.08447265625,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/max_abs":0.0091552734375,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/max_abs":0.09033203125,"train/train/tensor_act_model_layers_7_self_attn_k_proj/mean":-0.0006618350744247435,"train/train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/norm":259.09118251663875,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_7/grad/mean":-5.478754539010747e-07,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/norm":0.014999692337536678,"train/train/tensor_act_model_layers_0_input_layernorm/norm":9146.768310589385,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_2_post_attention_layernorm/norm":9158.621521007424,"train/train/tensor_act_model_layers_7_self_attn_o_proj/std":0.02723244874144052,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_norm/mean":-0.0038902759552001953,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/max_abs":0.07958984375,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_input_layernorm/max_abs":5.59375,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp_up_proj/mean":0.001192331314086914,"train/train/tensor_act_model_layers_3_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_8_input_layernorm_weight/max_abs":1,"train/train/global/grad/max_abs":0.185546875,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_input_layernorm/mean":0.006045341491699219,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/std":0.00017868155710299555,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/max_abs":0.08349609375,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/norm":2.129669444622657,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2/norm":1371.506467834684,"train/train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_k_proj/mean":-0.011730194091796875,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/norm":1.0230935953082407,"train/train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/norm":0.002631145786676072,"train/train/tensor_param_model_layers_3_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_4/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_v_proj/std":0.22168016473050778,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/global/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_norm_weight/std":0.0009002336688778526,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/norm":0.6655192832990424,"train/train/tensor_act_model_layers_3_input_layernorm/max_abs":4.75,"train/train/tensor_act_model_layers_0_mlp_up_proj/mean":-7.581710815429688e-05,"train/train/layer__model_layers_3/param/norm":17.937513610622013,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/mean":1.6309786587953568e-07,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/mean":-4.682460712501779e-07,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/mean":2.86102294921875e-05,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/mean":4.525645636022091e-09,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_norm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model/mean":-0.0038902759552001953,"train/train/layer_model_layers_4/grad/max_abs":0.009765625,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/std":0.001065419855838082,"train/train/tensor_act_model_layers_3_self_attn_o_proj/std":0.02468457451209307,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/std":7.5000117007145865e-06,"train/train/layer_model_layers_3/grad/mean":2.2994056518382846e-06,"train/train/tensor_act_model_layers_1_mlp/norm":783.6076718471597,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/max_abs":0.00604248046875,"train/train/tensor_act_model_layers_5/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/mean":0.00031280517578125,"train/train/tensor_act_model_layers_2_self_attn_o_proj/std":0.024912951683477253,"train/train/layer_model_layers_2/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn/norm":226.46827816100497,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp/max_abs":0.416015625,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/mean":-2.8014183044433594e-05,"train/train/tensor_act_model_layers_0_self_attn/norm":78.02399889508851,"train/train/tensor_act_model_layers_0_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/norm":4.4375,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_up_proj/mean":-0.0061511993408203125,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/norm":2053.7752410220805,"train/train/layer_model_layers_7/act/frac_near_user_limit":0,"train/train/global/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_4_self_attn_v_proj/norm":2030.936626263437,"train/train/global/act/std":0.4056012154736404,"train/train/tensor_act_model_layers_6_self_attn_o_proj/max_abs":0.2177734375,"train/train/tensor_act_model_layers_0_post_attention_layernorm/norm":9148.850097786815,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/std":7.781938305002603e-06,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/mean":9.324867278337479e-06,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_4_self_attn_k_proj/max_abs":1.1953125,"train/train/tensor_act_model_layers_7_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/mean":-3.5921111702919006e-06,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_8_mlp_down_proj/std":0.08425981926404569,"train/train/tensor_act_model_layers_2_post_attention_layernorm/std":1.0000019553098334,"train/train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/norm":0.9643301272382381,"train/train/tensor_act_model_layers_5_mlp_down_proj/max_abs":0.439453125,"train/train/layer_model_layers_5/act/max_abs":5.21875,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_lm_head/mean":-0.0027561187744140625,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_5_mlp_down_proj/norm":761.6220634845612,"train/train/tensor_act_model_layers_6_input_layernorm/mean":0.007315635681152344,"train/train/tensor_act_model_layers_2_self_attn_o_proj/norm":228.47226970029095,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/std":3.248117562247807e-05,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_0_self_attn_v_proj/mean":-0.0013871192932128906,"train/train/tensor_act_model_layers_6_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_k_proj/norm":2082.805685076487,"train/train/tensor_act_model_layers_0_mlp_down_proj/norm":771.7085156076147,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_7_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/max_abs":0.2578125,"train/train/tensor_act_model_layers_7_self_attn_k_proj/std":0.22613651179925,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/max_abs":0.00775146484375,"train/train/tensor_act_model_layers_2_post_attention_layernorm/mean":0.021453857421875,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/mean":1.928955316543579e-05,"train/train/tensor_param_model_layers_4_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/std":0.0006708584204051948,"train/train/tensor_act_model_layers_3/max_abs":0.95703125,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_6/act/max_abs":5.15625,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/max_abs":0.08154296875,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/max_abs":0.08642578125,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_dtype_limit":0} \ No newline at end of file diff --git a/wandb/run-20260809_061951-oe9tdw54/logs/debug-core.log b/wandb/run-20260809_061951-oe9tdw54/logs/debug-core.log new file mode 100644 index 0000000000000000000000000000000000000000..91527929b1102e9091c64062d69a77dbdcfcd197 --- /dev/null +++ b/wandb/run-20260809_061951-oe9tdw54/logs/debug-core.log @@ -0,0 +1,17 @@ +{"time":"2026-08-09T06:19:51.729724031Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmpb_g17z51/port-72015.txt","pid":72015,"detached":false,"idle-timeout":600000000000,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false} +{"time":"2026-08-09T06:19:51.730311525Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":72015} +{"time":"2026-08-09T06:19:51.730298315Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-72015-73483-908870555/socket","Net":"unix"}} +{"time":"2026-08-09T06:19:51.90660427Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"} +{"time":"2026-08-09T06:19:51.994419505Z","level":"INFO","msg":"handleInformInit: received","streamId":"oe9tdw54","id":"1(@)"} +{"time":"2026-08-09T06:19:52.269288272Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"oe9tdw54","id":"1(@)"} +{"time":"2026-08-09T06:19:57.589795151Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"fbcuqoz887gh"} +{"time":"2026-08-09T06:21:30.335906843Z","level":"INFO","msg":"connection: cancelling request","id":"1(@)","requestId":"fbcuqoz887gh"} +{"time":"2026-08-09T06:21:30.74790277Z","level":"INFO","msg":"processOutgoingData: finished","id":"1(@)"} +{"time":"2026-08-09T06:21:30.747914197Z","level":"INFO","msg":"connection: closing","id":"1(@)"} +{"time":"2026-08-09T06:21:30.747997951Z","level":"INFO","msg":"connection: closed successfully","id":"1(@)"} +{"time":"2026-08-09T06:21:30.748002644Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"1(@)"} +{"time":"2026-08-09T06:21:30.830098494Z","level":"INFO","msg":"server: parent process exited, terminating service process"} +{"time":"2026-08-09T06:21:30.83017242Z","level":"INFO","msg":"server: is shutting down"} +{"time":"2026-08-09T06:21:30.830421404Z","level":"INFO","msg":"server: forced shutdown"} +{"time":"2026-08-09T06:21:30.830353951Z","level":"INFO","msg":"server: listener closed","addr":{"Name":"/tmp/wandb-72015-73483-908870555/socket","Net":"unix"}} +{"time":"2026-08-09T06:21:30.830457345Z","level":"ERROR","msg":"main: Serve() returned error","error":"forced shutdown"} diff --git a/wandb/run-20260809_061951-oe9tdw54/logs/debug-internal.log b/wandb/run-20260809_061951-oe9tdw54/logs/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..4bcbb75125e3fdfb76887330a946575197f8cbc6 --- /dev/null +++ b/wandb/run-20260809_061951-oe9tdw54/logs/debug-internal.log @@ -0,0 +1,27 @@ +{"time":"2026-08-09T06:19:51.994617915Z","level":"INFO","msg":"wandb-core"} +{"time":"2026-08-09T06:19:51.995009066Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"} +{"time":"2026-08-09T06:19:52.269115339Z","level":"INFO","msg":"stream: created new stream","id":"oe9tdw54"} +{"time":"2026-08-09T06:19:52.269198637Z","level":"INFO","msg":"handler: started"} +{"time":"2026-08-09T06:19:52.269282318Z","level":"INFO","msg":"stream: started"} +{"time":"2026-08-09T06:19:52.269304675Z","level":"INFO","msg":"writer: started","stream_id":"oe9tdw54"} +{"time":"2026-08-09T06:19:52.269337129Z","level":"INFO","msg":"sender: started"} +{"time":"2026-08-09T06:19:53.747696701Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1} +{"time":"2026-08-09T06:19:53.854738149Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:20:08.747985212Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":0,"events_lines":2,"console_offset":1,"console_lines":4,"uploaded_len":2} +{"time":"2026-08-09T06:20:08.911257515Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:20:23.748068278Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":2,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T06:20:23.856021055Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:20:35.927162523Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":81} +{"time":"2026-08-09T06:20:35.960432275Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1678} +{"time":"2026-08-09T06:20:38.750260199Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":1,"events_offset":4,"events_lines":2,"console_offset":4,"console_lines":2} +{"time":"2026-08-09T06:20:38.978671285Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:20:53.748024569Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":6,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T06:20:53.914125364Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:21:08.748194781Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":8,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T06:21:08.930333319Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:21:23.751259648Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":1,"history_lines":1,"events_offset":10,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T06:21:23.933408958Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T06:21:30.747993636Z","level":"ERROR","msg":"PUT https://storage.googleapis.com/wandb-production.appspot.com/deepnevro-deepnevro/huggingface/oe9tdw54/config.yaml?X-Goog-Algorithm=GOOG4-RSA-SHA256&X-Goog-Credential=gorilla-files-url-signer%40wandb-production.iam.gserviceaccount.com%2F20260809%2Fauto%2Fstorage%2Fgoog4_request&X-Goog-Date=20260809T062130Z&X-Goog-Expires=86399&X-Goog-Signature=4c38a1e1aed34aa46bb1456c855b31828fbfac598138a4a6a6c2c9b1d5085b15762f21da7ce761c50c56979bf6db1e7b2655122604f989d0d75431d0fb903fd805b6856f5bf2f205a68aa286f90f0657d029944f9e1d96aff7a50db0f9d13e8f0612e7481da32e1dd6f460830638a36d0fbcc181eef2926945608b09f9796b9ee9dbae05496c53bf9e85b107bae0cba6b62883447f2e782d0266b663caa51c85c4440cf6e353980ac2a2e66a2ca31714d071f72b1cfc9eab515a697a3d908e5b68fc2bdfa9868b418cf83a521ce0f406a5a6600cf62863728c8623cd438f90031f85b2214bd211bbd5a1a9b6d01411febd870977881d73a62457c15c49cd02a2&X-Goog-SignedHeaders=host&X-User=deepnevro giving up after 1 attempt(s): context canceled","task":"DefaultUploadTask{FileKind: 1, Path: /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_061951-oe9tdw54/files/config.yaml, Name: config.yaml, Url: https://storage.googleapis.com/wandb-production.appspot.com/deepnevro-deepnevro/huggingface/oe9tdw54/config.yaml?X-Goog-Algorithm=GOOG4-RSA-SHA256&X-Goog-Credential=gorilla-files-url-signer%40wandb-production.iam.gserviceaccount.com%2F20260809%2Fauto%2Fstorage%2Fgoog4_request&X-Goog-Date=20260809T062130Z&X-Goog-Expires=86399&X-Goog-Signature=4c38a1e1aed34aa46bb1456c855b31828fbfac598138a4a6a6c2c9b1d5085b15762f21da7ce761c50c56979bf6db1e7b2655122604f989d0d75431d0fb903fd805b6856f5bf2f205a68aa286f90f0657d029944f9e1d96aff7a50db0f9d13e8f0612e7481da32e1dd6f460830638a36d0fbcc181eef2926945608b09f9796b9ee9dbae05496c53bf9e85b107bae0cba6b62883447f2e782d0266b663caa51c85c4440cf6e353980ac2a2e66a2ca31714d071f72b1cfc9eab515a697a3d908e5b68fc2bdfa9868b418cf83a521ce0f406a5a6600cf62863728c8623cd438f90031f85b2214bd211bbd5a1a9b6d01411febd870977881d73a62457c15c49cd02a2&X-Goog-SignedHeaders=host&X-User=deepnevro, Size: 9942}"} +{"time":"2026-08-09T06:21:30.748335068Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"} +{"time":"2026-08-09T06:21:30.749627405Z","level":"INFO","msg":"filestream: sending request","total_files":2,"console_offset":4,"console_lines":1,"uploaded_len":2} +{"time":"2026-08-09T06:21:30.74968725Z","level":"ERROR+4","msg":"filestream: fatal error: filestream: error making HTTP request: POST https://api.wandb.ai/files/deepnevro-deepnevro/huggingface/oe9tdw54/file_stream giving up after 1 attempt(s): context canceled. got response: "} diff --git a/wandb/run-20260809_061951-oe9tdw54/logs/debug.log b/wandb/run-20260809_061951-oe9tdw54/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..50928bee444189d45b493e9bd4f459317206d2a0 --- /dev/null +++ b/wandb/run-20260809_061951-oe9tdw54/logs/debug.log @@ -0,0 +1,26 @@ +2026-08-09 06:19:51,992 INFO MainThread:72015 [wandb_setup.py:_flush():81] Current SDK version is 0.28.1 +2026-08-09 06:19:51,992 INFO MainThread:72015 [wandb_setup.py:_flush():81] Configure stats pid to 72015 +2026-08-09 06:19:51,992 INFO MainThread:72015 [wandb_setup.py:_flush():81] Loading settings from environment variables +2026-08-09 06:19:51,992 INFO MainThread:72015 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_061951-oe9tdw54/logs/debug.log +2026-08-09 06:19:51,992 INFO MainThread:72015 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_061951-oe9tdw54/logs/debug-internal.log +2026-08-09 06:19:51,992 INFO MainThread:72015 [wandb_init.py:init():772] calling init triggers +2026-08-09 06:19:51,992 INFO MainThread:72015 [wandb_init.py:init():777] wandb.init called with sweep_config: {} +config: {'_wandb': {}} +2026-08-09 06:19:51,992 INFO MainThread:72015 [wandb_init.py:init():820] starting backend +2026-08-09 06:19:51,992 INFO MainThread:72015 [wandb_init.py:init():835] sending inform_init request +2026-08-09 06:19:52,269 INFO MainThread:72015 [wandb_init.py:init():840] backend started and connected +2026-08-09 06:19:52,271 INFO MainThread:72015 [wandb_init.py:init():910] updated telemetry +2026-08-09 06:19:52,277 INFO MainThread:72015 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout +2026-08-09 06:19:52,512 INFO MainThread:72015 [wandb_init.py:init():978] starting run threads in backend +2026-08-09 06:19:52,586 INFO MainThread:72015 [wandb_run.py:_console_start():2621] atexit reg +2026-08-09 06:19:52,586 INFO MainThread:72015 [wandb_run.py:_redirect():2471] redirect: wrap_raw +2026-08-09 06:19:52,586 INFO MainThread:72015 [wandb_run.py:_redirect():2540] Wrapping output streams. +2026-08-09 06:19:52,586 INFO MainThread:72015 [wandb_run.py:_redirect():2563] Redirects installed. +2026-08-09 06:19:52,589 INFO MainThread:72015 [wandb_init.py:init():1016] run started, returning control to user process +2026-08-09 06:19:52,590 INFO MainThread:72015 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.15.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 9, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'mlp', 'activation': 'tanh', 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'outio/mlp-tanh-9L_run', 'per_device_train_batch_size': 80, 'num_train_epochs': 1, 'max_steps': 750, 'learning_rate': 0.0005, 'lr_scheduler_type': 'constant', 'lr_scheduler_kwargs': None, 'warmup_steps': 0, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.01, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 16, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-mlp-tanh-9L-2.0M-20260809-061950', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 50, 'eval_delay': 0, 'per_device_eval_batch_size': 80, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': True, 'hub_token': '', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/finale-mlp-tanh-9L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1} +2026-08-09 06:19:52,591 INFO MainThread:72015 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 2001280 - > +2026-08-09 06:19:52,591 INFO MainThread:72015 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 2001280 None +2026-08-09 06:21:30,334 INFO MainThread:72015 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/huggingface/oe9tdw54 +2026-08-09 06:21:30,335 INFO MainThread:72015 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0 +2026-08-09 06:21:30,335 INFO MainThread:72015 [wandb_run.py:_restore():2570] restore +2026-08-09 06:21:30,335 INFO MainThread:72015 [wandb_run.py:_restore():2576] restore done diff --git a/wandb/run-20260809_061951-oe9tdw54/run-oe9tdw54.wandb b/wandb/run-20260809_061951-oe9tdw54/run-oe9tdw54.wandb new file mode 100644 index 0000000000000000000000000000000000000000..bd564e71e42b2f704cfe0d542e69c522ea60e13e --- /dev/null +++ b/wandb/run-20260809_061951-oe9tdw54/run-oe9tdw54.wandb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c7a7ba5c0c8fccb841e81f05aee62ad92328b8179099236bd896e0f63d4dc386 +size 622592 diff --git a/wandb/run-20260809_070213-uvqyddz0/files/config.yaml b/wandb/run-20260809_070213-uvqyddz0/files/config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..d0b2b77cd85c76ac85cfc74bdd878e9399698eb7 --- /dev/null +++ b/wandb/run-20260809_070213-uvqyddz0/files/config.yaml @@ -0,0 +1,432 @@ +_name_or_path: + value: "" +_wandb: + value: + cli_version: 0.28.1 + e: + 8ze12vi8sgbkef1obbpgztpskrwef8xc: + args: + - --config + - configs/baseline.yaml + - --variants + - glu-tanh-94L + - --push + codePath: sweep.py + codePathLocal: sweep.py + cpu_count: 112 + cpu_count_logical: 224 + cudaVersion: "12.4" + disk: + /: + total: "1560765693952" + used: "708283842560" + email: deepnevro@gmail.com + executable: /mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python + git: + commit: 34b8d2e8f9a0c5751333310e69fa0c1056381deb + remote: https://github.com/deepnevro/Activation.git + gpu: NVIDIA H100 80GB HBM3 + gpu_count: 8 + gpu_nvidia: + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-39c684a5-fde6-83d7-1663-0859795881ae + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-132944c4-b689-2b5f-89a4-d730401677ab + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-bc6c3e3c-9b90-09ca-c034-774961847c54 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9 + - architecture: Hopper + cudaCores: 16896 + memoryTotal: "85520809984" + name: NVIDIA H100 80GB HBM3 + uuid: GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea + host: deeplens-k3s-node1 + memory: + total: "2164089937920" + os: Linux-5.15.0-126-generic-x86_64-with-glibc2.35 + program: /mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py + python: CPython 3.11.15 + root: /mnt/data/zainulabideen/zain-exp/notebooks/Activation + startedAt: "2026-08-09T07:02:13.952337Z" + writerId: 8ze12vi8sgbkef1obbpgztpskrwef8xc + m: + - "1": train/global_step + "6": + - 3 + "7": [] + - "2": '*' + "5": 1 + "6": + - 1 + "7": [] + python_version: 3.11.15 + t: + "1": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "2": + - 1 + - 5 + - 11 + - 41 + - 49 + - 51 + - 53 + - 71 + "3": + - 2 + - 7 + - 13 + - 19 + - 41 + - 66 + "4": 3.11.15 + "5": 0.28.1 + "6": 5.15.0.dev0 + "9": + "1": transformers_trainer + "12": 0.28.1 + "13": linux-x86_64 +accelerator_config: + value: + dispatch_batches: null + even_batches: true + gradient_accumulation_kwargs: null + non_blocking: false + split_batches: false + use_seedable_sampler: true +activation: + value: tanh +adam_beta1: + value: 0.9 +adam_beta2: + value: 0.999 +adam_epsilon: + value: 1e-08 +architectures: + value: null +attention_bias: + value: false +attention_dropout: + value: 0 +auto_find_batch_size: + value: false +average_tokens_across_devices: + value: true +batch_eval_metrics: + value: false +bf16: + value: true +bf16_full_eval: + value: false +bos_token_id: + value: 1 +chunk_size_feed_forward: + value: 0 +data_seed: + value: 42 +dataloader_drop_last: + value: false +dataloader_in_order: + value: true +dataloader_multiprocessing_context: + value: null +dataloader_num_workers: + value: 0 +dataloader_persistent_workers: + value: false +dataloader_pin_memory: + value: true +dataloader_prefetch_factor: + value: null +ddp_backend: + value: null +ddp_broadcast_buffers: + value: null +ddp_bucket_cap_mb: + value: null +ddp_find_unused_parameters: + value: null +ddp_static_graph: + value: null +ddp_timeout: + value: 1800 +debug: + value: [] +deepspeed: + value: null +disable_tqdm: + value: false +do_eval: + value: true +do_predict: + value: false +do_train: + value: false +dtype: + value: null +enable_jit_checkpoint: + value: false +eos_token_id: + value: 2 +eval_accumulation_steps: + value: null +eval_delay: + value: 0 +eval_do_concat_batches: + value: true +eval_on_start: + value: false +eval_steps: + value: 50 +eval_strategy: + value: steps +eval_use_gather_object: + value: false +fp16: + value: false +fp16_full_eval: + value: false +fsdp: + value: null +fsdp_config: + value: null +full_determinism: + value: false +gradient_accumulation_steps: + value: 4 +gradient_checkpointing: + value: false +gradient_checkpointing_kwargs: + value: null +greater_is_better: + value: null +head_dim: + value: 32 +hidden_act: + value: silu +hidden_size: + value: 128 +hub_always_push: + value: false +hub_model_id: + value: w-ahmad/A-glu-tanh-94L +hub_private_repo: + value: null +hub_revision: + value: null +hub_strategy: + value: every_save +hub_token: + value: +id2label: + value: + "0": LABEL_0 + "1": LABEL_1 +ignore_data_skip: + value: false +include_for_metrics: + value: [] +include_num_input_tokens_seen: + value: "no" +initializer_range: + value: 0.02 +intermediate_size: + value: 256 +is_encoder_decoder: + value: false +label_names: + value: null +label_smoothing_factor: + value: 0 +label2id: + value: + LABEL_0: 0 + LABEL_1: 1 +learning_rate: + value: 0.001 +length_column_name: + value: length +liger_kernel_config: + value: null +load_best_model_at_end: + value: false +local_rank: + value: -1 +log_level: + value: passive +log_level_replica: + value: warning +log_on_each_node: + value: true +logging_first_step: + value: false +logging_nan_inf_filter: + value: true +logging_steps: + value: 20 +logging_strategy: + value: steps +lr_scheduler_kwargs: + value: null +lr_scheduler_type: + value: constant +max_grad_norm: + value: 1 +max_position_embeddings: + value: 512 +max_steps: + value: 1500 +metric_for_best_model: + value: null +mlp_bias: + value: false +mlp_type: + value: glu +model/num_parameters: + value: 15949440 +model_type: + value: tiny_llama +neftune_noise_alpha: + value: null +num_attention_heads: + value: 4 +num_hidden_layers: + value: 94 +num_key_value_heads: + value: 4 +num_train_epochs: + value: 1 +optim: + value: adamw_torch_fused +optim_args: + value: null +optim_target_modules: + value: null +output_attentions: + value: false +output_dir: + value: out/glu-tanh-94L_run +output_hidden_states: + value: false +pad_token_id: + value: 0 +parallelism_config: + value: null +per_device_eval_batch_size: + value: 128 +per_device_train_batch_size: + value: 128 +prediction_loss_only: + value: false +pretraining_tp: + value: 1 +problem_type: + value: null +project: + value: huggingface +push_to_hub: + value: true +remove_unused_columns: + value: false +report_to: + value: + - wandb +restore_callback_states_from_checkpoint: + value: false +resume_from_checkpoint: + value: null +return_dict: + value: true +rms_norm_eps: + value: 1e-06 +rope_parameters: + value: + rope_theta: 10000 + rope_type: default +run_name: + value: LM-glu-tanh-94L-15.9M-20260809-070212 +save_on_each_node: + value: false +save_only_model: + value: false +save_steps: + value: 100 +save_strategy: + value: steps +save_total_limit: + value: null +seed: + value: 42 +skip_memory_metrics: + value: true +tf32: + value: null +tie_word_embeddings: + value: true +tokenizer_name: + value: w-ahmad/tiny-stories-tokenizer +torch_compile: + value: false +torch_compile_backend: + value: null +torch_compile_mode: + value: null +torch_empty_cache_steps: + value: null +trackio_bucket_id: + value: null +trackio_space_id: + value: null +trackio_static_space_id: + value: null +train_sampling_strategy: + value: random +transformers_version: + value: 5.15.0.dev0 +use_cache: + value: false +use_cpu: + value: false +use_liger_kernel: + value: false +vocab_size: + value: 4096 +warmup_steps: + value: 0 +weight_decay: + value: 0.01 diff --git a/wandb/run-20260809_070213-uvqyddz0/files/output.log b/wandb/run-20260809_070213-uvqyddz0/files/output.log new file mode 100644 index 0000000000000000000000000000000000000000..0239bb510b7f9084bcc6ebe0b729ab229bbda0e8 --- /dev/null +++ b/wandb/run-20260809_070213-uvqyddz0/files/output.log @@ -0,0 +1,7 @@ +[transformers] `use_return_dict` is deprecated! Use `return_dict` instead! +[INFO] Causal mask (float with -inf) applied to all attention layers. +/mnt/data/zainulabideen/zain-exp/notebooks/Activation/exp.py:487: UserWarning: std(): degrees of freedom is <= 0. Correction should be strictly less than the reduction factor (input numel divided by output numel). (Triggered internally at /pytorch/aten/src/ATen/native/ReduceOps.cpp:1831.) + "std": tensor.std().item(), + 3%|█ | 42/1500 [01:24<46:06, 1.90s/it] +{'loss': '28.46', 'grad_norm': '3.625', 'learning_rate': '0.001', 'epoch': '0.01078', 'train/total_time_seconds': '33.54', 'train/time_per_step_avg': '1.677', 'train/epoch_time_elapsed': '41.83', 'train/estimated_remaining_minutes': '41.36', 'train/global/act/norm': '8.913e+04', 'train/global/act/mean': '-0.002626', 'train/global/act/std': '0.4187', 'train/global/act/max_abs': '8.323', 'train/global/act/frac_near_dtype_limit': '0', 'train/global/act/frac_near_user_limit': '0', 'train/global/grad/norm': '3.594', 'train/global/grad/mean': '-7.644e-08', 'train/global/grad/std': '0.0004499', 'train/global/grad/max_abs': '0.126', 'train/global/grad/frac_near_dtype_limit': '0', 'train/global/grad/frac_near_user_limit': '0', 'train/global/param/norm': '174.8', 'train/global/param/mean': '0.001521', 'train/global/param/std': '0.04376', 'train/global/param/max_abs': '1', 'train/global/param/frac_near_dtype_limit': '0', 'train/global/param/frac_near_user_limit': '0', 'train/layer_model_layers_73/act/norm': '9254', 'train/layer_model_layers_73/act/mean': '0.0007436', 'train/layer_model_layers_73/act/std': '0.427', 'train/layer_model_layers_73/act/max_abs': '4.531', 'train/layer_model_layers_73/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_73/act/frac_near_user_limit': '0', 'train/layer_model_layers_73/grad/norm': '0.1576', 'train/layer_model_layers_73/grad/mean': '1.061e-08', 'train/layer_model_layers_73/grad/std': '0.0001946', 'train/layer_model_layers_73/grad/max_abs': '0.004974', 'train/layer_model_layers_73/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_73/grad/frac_near_user_limit': '0', 'train/layer_model_layers_82/act/norm': '9293', 'train/layer_model_layers_82/act/mean': '1.821e-05', 'train/layer_model_layers_82/act/std': '0.4288', 'train/layer_model_layers_82/act/max_abs': '4.781', 'train/layer_model_layers_82/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_82/act/frac_near_user_limit': '0', 'train/layer_model_layers_82/grad/norm': '0.1477', 'train/layer_model_layers_82/grad/mean': '1.225e-07', 'train/layer_model_layers_82/grad/std': '0.0001823', 'train/layer_model_layers_82/grad/max_abs': '0.00386', 'train/layer_model_layers_82/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_82/grad/frac_near_user_limit': '0', 'train/layer_model_layers_14/act/norm': '8970', 'train/layer_model_layers_14/act/mean': '-0.008702', 'train/layer_model_layers_14/act/std': '0.4143', 'train/layer_model_layers_14/act/max_abs': '5.062', 'train/layer_model_layers_14/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_14/act/frac_near_user_limit': '0', 'train/layer_model_layers_14/grad/norm': '0.3804', 'train/layer_model_layers_14/grad/mean': '1.381e-06', 'train/layer_model_layers_14/grad/std': '0.0004696', 'train/layer_model_layers_14/grad/max_abs': '0.009155', 'train/layer_model_layers_14/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_14/grad/frac_near_user_limit': '0', 'train/layer__model_layers_86/param/norm': '17.93', 'train/layer__model_layers_86/param/mean': '0.001569', 'train/layer__model_layers_86/param/std': '0.04425', 'train/layer__model_layers_86/param/max_abs': '1', 'train/layer__model_layers_86/param/frac_near_dtype_limit': '0', 'train/layer__model_layers_86/param/frac_near_user_limit': '0', 'train/layer_model_layers_61/act/norm': '9186', 'train/layer_model_layers_61/act/mean': '0.0001298', 'train/layer_model_layers_61/act/std': '0.4239', 'train/layer_model_layers_61/act/max_abs': '4.656', 'train/layer_model_layers_61/act/frac_near_dtype_limit': '0', 'train/layer_model_layers_61/act/frac_near_user_limit': '0', 'train/layer_model_layers_61/grad/norm': '0.1668', 'train/layer_model_layers_61/grad/mean': '-6.588e-08', 'train/layer_model_layers_61/grad/std': '0.0002058', 'train/layer_model_layers_61/grad/max_abs': '0.004852', 'train/layer_model_layers_61/grad/frac_near_dtype_limit': '0', 'train/layer_model_layers_61/grad/frac_near_user_limit': '0', 'train/layer_model_layers_58/act/norm': '9170', 'train/layer_model_layers_58/act/mean': '-0.0007553', 'train/layer_model_layers_5 +{'loss': '24.26', 'grad_norm': '0.2412', 'learning_rate': '0.001', 'epoch': '0.02157', 'train/total_time_seconds': '63.84', 'train/time_per_step_avg': '1.596', 'train/epoch_time_elapsed': '80.15', 'train/estimated_remaining_minutes': '38.84'} diff --git a/wandb/run-20260809_070213-uvqyddz0/files/requirements.txt b/wandb/run-20260809_070213-uvqyddz0/files/requirements.txt new file mode 100644 index 0000000000000000000000000000000000000000..123b15ebf857859624f7f4332e92341c8ef13fdf --- /dev/null +++ b/wandb/run-20260809_070213-uvqyddz0/files/requirements.txt @@ -0,0 +1,149 @@ +asttokens==3.0.1 +comm==0.2.3 +debugpy==1.8.21 +decorator==5.3.1 +executing==2.2.1 +nest-asyncio==1.6.0 +parso==0.8.7 +platformdirs==4.11.0 +psutil==7.2.2 +ptyprocess==0.7.0 +pure_eval==0.2.3 +Pygments==2.20.0 +pyzmq==27.1.0 +setuptools==83.0.0 +six==1.17.0 +tornado==6.5.7 +traitlets==5.15.0 +fsspec==2026.4.0 +wcwidth==0.8.2 +ipython_pygments_lexers==1.1.1 +jedi==0.20.0 +jupyter_core==5.9.1 +matplotlib-inline==0.2.2 +pexpect==4.9.0 +prompt_toolkit==3.0.53 +python-dateutil==2.9.0.post0 +stack_data==0.6.3 +wheel==0.47.0 +jupyter_client==8.9.1 +pip==26.1.2 +ipython==9.15.0 +ipykernel==7.2.0 +threadpoolctl==3.6.0 +pyparsing==3.3.2 +typing_extensions==4.15.0 +Jinja2==3.1.6 +narwhals==2.24.0 +kiwisolver==1.5.0 +joblib==1.5.3 +fonttools==4.63.0 +cycler==0.12.1 +scipy==1.17.1 +pandas==3.0.5 +contourpy==1.3.3 +scikit-learn==1.9.0 +matplotlib==3.11.1 +urllib3==2.7.0 +tqdm==4.70.0 +idna==3.18 +charset-normalizer==3.4.9 +certifi==2026.7.22 +requests==2.34.2 +seaborn==0.13.2 +uv==0.12.0 +shellingham==1.5.4 +mpmath==1.3.0 +attrs==26.1.0 +hf-xet==1.5.2 +nvidia-nccl-cu12==2.21.5 +MarkupSafe==3.0.3 +regex==2026.7.19 +importlib_metadata==9.0.0 +httpcore==1.0.9 +annotated-doc==0.0.5 +multidict==6.7.1 +aiohttp==3.14.3 +aiosignal==1.4.0 +xxhash==3.8.1 +aiohappyeyeballs==2.7.1 +mdurl==0.1.2 +cuda-toolkit==13.0.3.0 +networkx==3.6.1 +PyYAML==6.0.3 +nvidia-cufile==1.15.1.6 +typer==0.27.0 +torchaudio==2.6.0+cu124 +rich==15.0.0 +nvidia-cufft-cu12==11.2.1.3 +h11==0.16.0 +dill==0.4.1 +cuda-pathfinder==1.6.0 +filelock==3.29.0 +nvidia-nvtx-cu12==12.4.127 +httpx==0.28.1 +anyio==4.14.2 +numpy==2.4.4 +yarl==1.24.5 +click==8.4.2 +triton==3.2.0 +frozenlist==1.8.0 +zipp==4.1.0 +propcache==0.5.2 +tokenizers==0.22.2 +markdown-it-py==4.2.0 +nvidia-cuda-runtime==13.0.96 +cuda-bindings==13.3.1 +nvidia-cuda-cupti==13.0.85 +torch==2.6.0+cu124 +multiprocess==0.70.19 +pillow==12.2.0 +transformers==5.15.0.dev0 +wandb==0.28.1 +nvidia-curand==10.4.0.35 +sympy==1.13.1 +nvidia-cusparse==12.6.3.3 +nvidia-cuda-nvrtc==13.0.88 +typing-inspection==0.4.2 +nvidia-cusolver==12.0.4.66 +nvidia-cufft==12.0.0.61 +nvidia-cudnn-cu13==9.20.0.48 +nvidia-cublas==13.1.1.3 +pyarrow==25.0.0 +evaluate==0.4.6 +diffusers==0.39.0 +pydantic==2.13.4 +annotated-types==0.8.0 +protobuf==7.35.1 +sentry-sdk==2.66.1 +einops==0.8.2 +packaging==26.2 +nvidia-nvjitlink-cu12==12.4.127 +nvidia-curand-cu12==10.3.5.147 +nvidia-cusparselt-cu12==0.6.2 +nvidia-cusparse-cu12==12.3.1.170 +nvidia-cuda-runtime-cu12==12.4.127 +torchvision==0.21.0+cu124 +nvidia-cuda-nvrtc-cu12==12.4.127 +nvidia-cuda-cupti-cu12==12.4.127 +nvidia-cusolver-cu12==11.6.1.9 +nvidia-cublas-cu12==12.4.5.8 +nvidia-cudnn-cu12==9.1.0.70 +huggingface_hub==1.26.0 +datasets==5.0.1 +safetensors==0.8.0 +accelerate==1.14.0 +pydantic_core==2.46.4 +ninja==1.13.0 +autocommand==2.2.2 +backports.tarfile==1.2.0 +importlib_metadata==8.7.1 +jaraco.text==4.0.0 +jaraco.context==6.1.0 +jaraco.functools==4.4.0 +more-itertools==10.8.0 +packaging==26.0 +platformdirs==4.4.0 +tomli==2.4.0 +wheel==0.46.3 +zipp==3.23.0 diff --git a/wandb/run-20260809_070213-uvqyddz0/files/wandb-metadata.json b/wandb/run-20260809_070213-uvqyddz0/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..85abe6275ad842261b71b62495451a05e9c14b09 --- /dev/null +++ b/wandb/run-20260809_070213-uvqyddz0/files/wandb-metadata.json @@ -0,0 +1,96 @@ +{ + "os": "Linux-5.15.0-126-generic-x86_64-with-glibc2.35", + "python": "CPython 3.11.15", + "startedAt": "2026-08-09T07:02:13.952337Z", + "args": [ + "--config", + "configs/baseline.yaml", + "--variants", + "glu-tanh-94L", + "--push" + ], + "program": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation/sweep.py", + "codePath": "sweep.py", + "codePathLocal": "sweep.py", + "git": { + "remote": "https://github.com/deepnevro/Activation.git", + "commit": "34b8d2e8f9a0c5751333310e69fa0c1056381deb" + }, + "email": "deepnevro@gmail.com", + "root": "/mnt/data/zainulabideen/zain-exp/notebooks/Activation", + "host": "deeplens-k3s-node1", + "executable": "/mnt/data/zainulabideen/zain-exp/notebooks/my_env/bin/python", + "cpu_count": 112, + "cpu_count_logical": 224, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1560765693952", + "used": "708283842560" + } + }, + "memory": { + "total": "2164089937920" + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-39c684a5-fde6-83d7-1663-0859795881ae" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-68012e5a-38b6-b643-0ca6-62fb66720bf3" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-132944c4-b689-2b5f-89a4-d730401677ab" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-2df386cc-6d26-d0e2-7a2d-a057b0d95864" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bfa16575-1d94-1aa2-4537-2c93433f42ef" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-bc6c3e3c-9b90-09ca-c034-774961847c54" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-00a441e1-7c95-e7d6-4c35-43d6b291aea9" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper", + "uuid": "GPU-1c4d29a2-4647-6fce-d8fc-0c5ecfbbd6ea" + } + ], + "cudaVersion": "12.4", + "writerId": "8ze12vi8sgbkef1obbpgztpskrwef8xc" +} \ No newline at end of file diff --git a/wandb/run-20260809_070213-uvqyddz0/files/wandb-summary.json b/wandb/run-20260809_070213-uvqyddz0/files/wandb-summary.json new file mode 100644 index 0000000000000000000000000000000000000000..218fdec85c48a3fbf6ba2114c1cf0fdd989bd923 --- /dev/null +++ b/wandb/run-20260809_070213-uvqyddz0/files/wandb-summary.json @@ -0,0 +1 @@ +{"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_79/grad/max_abs":0.004425048828125,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/std":0.00010634611829458478,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/norm":0.002378206002006321,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/norm":0.1277105186652955,"train/train/tensor_act_model_layers_82_post_attention_layernorm/max_abs":4.75,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_70_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_36_self_attn_k_proj/max_abs":1.1484375,"train/train/layer_model_layers_0/grad/std":0.0015795309558291539,"train/train/tensor_act_model_layers_43_mlp_down_proj/mean":0.0009212493896484375,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/norm":0.10215994367109521,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/max_abs":0.0859375,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/std":0.00026823543461835055,"train/train/tensor_act_model_layers_75_mlp_up_proj/mean":0.002536773681640625,"train/train/tensor_act_model_layers_79_self_attn_q_proj/std":0.22119971127108054,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/mean":-2.6030465960502625e-07,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_21_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_25/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/mean":-1.8708306015469134e-09,"train/train/tensor_act_model_layers_10_self_attn/max_abs":0.2158203125,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/std":5.152055691600731e-07,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/norm":0.00012923430642057537,"train/train/layer_model_layers_3/grad/frac_near_user_limit":0,"train/train/layer__model_layers_11/param/std":0.044255564427576013,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/norm":0.06253646740363032,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/mean":3.62396240234375e-05,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/norm":0.002468507077240665,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/norm":0.03001769547318835,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_mlp_gate_proj_weight/std":8.24049857718573e-05,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_20/max_abs":1.078125,"train/train/tensor_param_model_layers_9_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_77/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp_gate_proj/std":0.2268088297538107,"train/train/tensor_act_model_layers_42_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/mean":-3.814697265625e-05,"train/train/tensor_act_model_layers_69_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_input_layernorm/norm":5792.592651369548,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/std":0.00016031243047486563,"train/train/tensor_param_model_layers_47_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/max_abs":0.0771484375,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/max_abs":0.08447265625,"train/train/tensor_act_model_layers_70_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_mlp_gate_proj_weight/mean":0.0001697540283203125,"train/train/tensor_act_model_layers_1_self_attn_k_proj/norm":1319.831802680977,"train/train/tensor_grad_model_layers_71_mlp_gate_proj_weight/mean":-1.0384246706962585e-07,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/max_abs":0.002685546875,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/max_abs":0.0012359619140625,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/max_abs":0.00115203857421875,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/std":0,"train/train/layer__model_layers_46/param/norm":17.935063532261324,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/max_abs":0.00010919570922851562,"train/train/tensor_act_model_layers_76_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/mean":2.872943878173828e-05,"train/train/tensor_act_model_layers_82_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_norm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_1_post_attention_layernorm/std":0.9980530710643881,"train/train/tensor_act_model_layers_0_self_attn_o_proj/mean":-0.000823974609375,"train/train/layer_model_layers_8/grad/max_abs":0.01104736328125,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/mean":7.772445678710938e-05,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/max_abs":0.07861328125,"train/train/tensor_param_model_layers_74_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_90/param/std":0.044251134436937546,"train/train/tensor_param_model_layers_77_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_25_self_attn_k_proj/norm":1287.8629533850572,"train/train/tensor_act_model_layers_46_self_attn_k_proj/max_abs":1.03125,"train/train/tensor_grad_model_layers_21_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp_gate_proj/mean":0.0026226043701171875,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/norm":0.05801949431576725,"train/train/tensor_act_model_layers_38_mlp_down_proj/norm":89.83683625483374,"train/train/tensor_act_model_layers_22_self_attn_o_proj/norm":272.33041138620234,"train/train/layer__model_layers_60/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_mlp_up_proj/max_abs":1.09375,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/std":8.310096354750486e-05,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/std":0.0009812412999521437,"train/train/tensor_act_model_layers_70_mlp_gate_proj/max_abs":1.1015625,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/mean":9.890645742416382e-07,"train/train/tensor_act_model_layers_19_input_layernorm/mean":-0.05242919921875,"train/train/tensor_act_model_layers_76_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/mean":-1.0861549526453018e-07,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/std":0.019775390625,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/norm":0.10300969700396835,"train/train/tensor_param_model_layers_66_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/mean":-1.6167759895324707e-05,"train/train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/mean":2.047419548034668e-05,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/mean":0.00010395050048828125,"train/train/layer_model_layers_49/act/norm":9128.944957935375,"train/train/tensor_act_model_layers_87_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_post_attention_layernorm/max_abs":4.6875,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_81_mlp_gate_proj/max_abs":1.203125,"train/train/tensor_act_model_layers_72_post_attention_layernorm/std":1.0000020380718675,"train/train/tensor_act_model_layers_70_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/max_abs":0.07958984375,"train/train/tensor_act_model_layers_16_self_attn/mean":0.00031375885009765625,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/mean":7.82012939453125e-05,"train/train/tensor_param_model_layers_80_mlp_gate_proj_weight/mean":0.00024318695068359375,"train/train/layer_model_layers_33/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8/norm":755.7953612580402,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_22/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/norm":0.037642019281656315,"train/train/tensor_act_model_layers_46/norm":1934.759627016494,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/mean":0.00012683868408203125,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/max_abs":0.000141143798828125,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/std":8.776204675394496e-05,"train/train/tensor_act_model_layers_14_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer__model_layers_44/param/max_abs":1,"train/train/tensor_grad_model_layers_41_mlp_gate_proj_weight/max_abs":0.00142669677734375,"train/train/tensor_act_model_layers_19_mlp_down_proj/mean":-0.00017303228378295898,"train/train/tensor_act_model_layers_59_mlp_gate_proj/max_abs":1.140625,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_59_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_gate_proj_weight/max_abs":0.0888671875,"train/train/tensor_act_model_layers_72_mlp_down_proj/std":0.015533686741760424,"train/train/tensor_act_model_layers_19/max_abs":1.0625,"train/train/tensor_act_model_layers_89_mlp_up_proj/norm":1855.2965193325995,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/mean":2.2112089936854318e-10,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/max_abs":0.0751953125,"train/train/tensor_act_model_layers_12/std":0.1645553316568179,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/norm":0.00011453177898222493,"train/train/tensor_act_model_layers_75_mlp/std":0.015839705856286513,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_65_self_attn_k_proj/std":0.22607870311826153,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/max_abs":0.00148773193359375,"train/train/tensor_act_model_layers_71_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp_down_proj/norm":90.31826523593222,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/mean":-1.16787850856781e-06,"train/train/tensor_act_model_layers_20_mlp/std":0.015930739665679237,"train/train/tensor_act_model_layers_59_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_mlp_down_proj/norm":93.71499610338645,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/norm":2.59375,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/mean":5.373731255531311e-07,"train/train/tensor_act_model_layers_59_mlp_up_proj/mean":-0.0012941360473632812,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/norm":0.08887422973899936,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/grad/max_abs":0.0123291015625,"train/train/tensor_act_model_layers_0_self_attn_q_proj/norm":1298.9157776486948,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/max_abs":0.0013275146484375,"train/train/tensor_param_model_layers_49_input_layernorm_weight/std":0,"train/train/layer__model_layers_6/param/norm":17.9274593522932,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/max_abs":0.0052490234375,"train/train/layer_model_layers_92/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/norm":0.0007977507033151282,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/max_abs":0.08447265625,"train/train/tensor_act_model_layers_11/norm":907.7799843186497,"train/train/tensor_act_model_layers_36/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/std":0.00012926750050141686,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/std":4.6532246809165674e-07,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/std":0.0005979290338987081,"train/train/tensor_act_model_layers_31_mlp_up_proj/norm":1866.3045129657482,"train/train/tensor_act_model_layers_3_input_layernorm/norm":5791.390014657087,"train/train/tensor_act_model_layers_1_self_attn_k_proj/std":0.22778376328938568,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_61/act/norm":9185.614820804056,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/norm":0.000960639315300451,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/max_abs":0.08203125,"train/train/tensor_act_model_layers_6_mlp_gate_proj/max_abs":1.15625,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/norm":0.0007737608223875711,"train/train/tensor_act_model_layers_71_input_layernorm/norm":5792.592651367984,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56/std":0.3681786540988311,"train/train/tensor_param_model_layers_9_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_37/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/mean":1.8905848264694214e-06,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/max_abs":0.07958984375,"train/train/tensor_act_model_layers_5_self_attn_q_proj/std":0.21875212872695574,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_90/mean":0.00562286376953125,"train/train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/mean":-0.00014972686767578125,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/mean":4.220008850097656e-05,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_o_proj/norm":270.2847000196818,"train/train/tensor_act_model_layers_28_input_layernorm/max_abs":4.65625,"train/train/tensor_act_model_layers_67_post_attention_layernorm/max_abs":4.40625,"train/train/tensor_act_model_layers_47_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/max_abs":0.0003223419189453125,"train/train/tensor_act_model_layers_5_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/mean":0.000133514404296875,"train/train/tensor_act_model_layers_54_mlp_gate_proj/max_abs":1.1328125,"train/train/tensor_act_model_layers_40_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/mean":-6.628036499023438e-05,"train/train/tensor_act_model_layers_45_mlp_gate_proj/max_abs":1.1484375,"train/train/tensor_act_model_layers_88_self_attn_v_proj/norm":1316.5630089138674,"train/train/tensor_act_model_layers_41_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/norm":0.03663451902352592,"train/train/tensor_act_model_layers_29_mlp_up_proj/mean":-0.009765625,"train/train/tensor_param_model_layers_62_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_38_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_q_proj/mean":0.004138946533203125,"train/train/tensor_act_model_layers_60_mlp/max_abs":0.0849609375,"train/train/tensor_act_model_layers_10_mlp/max_abs":0.08154296875,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/max_abs":0.08984375,"train/train/layer_model_layers_56/act/mean":-0.0015639747892107283,"train/train/tensor_grad_model_layers_18_mlp_gate_proj_weight/mean":-1.1918018572032452e-08,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_10_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/mean":-0.00010204315185546875,"train/train/tensor_act_model_layers_40/std":0.3115323430119914,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_post_attention_layernorm/norm":5792.567016605538,"train/train/tensor_act_model_layers_74_self_attn_o_proj/norm":288.28196579022557,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_35_mlp_gate_proj_weight/std":0.00012619981258531532,"train/train/tensor_act_model_layers_19_self_attn_k_proj/std":0.230231824394713,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/norm":0.07247112027824287,"train/train/tensor_act_model_layers_84_self_attn_q_proj/max_abs":1.078125,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/mean":-1.4491379261016846e-06,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_25_input_layernorm/norm":5792.55944824746,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/max_abs":5.078315734863281e-05,"train/train/tensor_act_model_layers_22_self_attn_k_proj/std":0.2163139756931919,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/max_abs":0.00012874603271484375,"train/train/tensor_act_model_layers_12_self_attn_o_proj/frac_near_user_limit":0,"train/train/total_time_seconds":63.8401404209435,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_mlp_down_proj/max_abs":0.0849609375,"train/train/tensor_act_model_layers_54_self_attn_k_proj/max_abs":1.03125,"train/train/layer_model_layers_84/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/max_abs":0.00110626220703125,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/max_abs":0.08251953125,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_41_mlp_gate_proj_weight/max_abs":0.08642578125,"train/train/tensor_act_model_layers_82_self_attn/std":0.04993066708292556,"train/train/tensor_act_model_layers_19_mlp_up_proj/mean":-0.022369384765625,"train/train/tensor_act_model_layers_51_self_attn_v_proj/mean":-0.015106201171875,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp/std":0.015503836343847308,"train/train/tensor_param_model_layers_90_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/max_abs":0.00421142578125,"train/train/layer_model_layers_52/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_o_proj/std":0.046449112958302216,"train/train/tensor_act_model_layers_86_mlp_up_proj/norm":1850.9478797609581,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/std":0.00013455774173031857,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/norm":0.013084783076137934,"train/train/tensor_param_model_layers_19_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_48_mlp_up_proj/norm":1825.1488767612466,"train/train/tensor_act_model_layers_34_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_mlp_gate_proj/max_abs":1.203125,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/mean":-4.678964614868164e-06,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/max_abs":0.0011444091796875,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/std":6.0984994370034945e-05,"train/train/tensor_act_model_layers_57_self_attn_q_proj/norm":1299.4763070058277,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/std":0.0201416015625,"train/train/layer_model_layers_85/grad/norm":0.14096787783616066,"train/train/tensor_act_model_layers_50_mlp/mean":-0.00026535987854003906,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/std":3.579736890910013e-05,"train/train/tensor_act_model_layers_42_mlp_up_proj/norm":1876.7063040814883,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/std":0.00012204736781397165,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn_q_proj/mean":0.0002956390380859375,"train/train/tensor_act_model_layers_15_post_attention_layernorm/mean":-0.0606689453125,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_22/grad/std":0.00034126426731159834,"train/train/layer_model_layers_17/act/mean":-0.010443006243024553,"train/train/tensor_act_model_layers_53_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/mean":1.1676456779241562e-07,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/norm":0.10920784302307543,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_45_self_attn_v_proj/std":0.22681084008301872,"train/train/layer_model_layers_50/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_q_proj/std":0.22778849342897348,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp/norm":90.82348488364937,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/max_abs":0.07861328125,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/mean":6.349291652441025e-07,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/std":8.099258167010827e-05,"train/train/layer_model_layers_66/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_self_attn/std":0.05035436493330927,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/max_abs":6.3478946685791016e-06,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/max_abs":0.0035552978515625,"train/train/layer_model_layers_60/act/std":0.42318041528594286,"train/train/layer_model_layers_31/grad/norm":0.23246575905579292,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/mean":0.00013828277587890625,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/std":8.527750131927965e-07,"train/train/tensor_act_model_layers_72_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_19_mlp_gate_proj_weight/mean":-1.9688159227371216e-06,"train/train/tensor_act_model_layers_55_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_input_layernorm/norm":5792.570922853613,"train/train/tensor_param_model_layers_55_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_mlp_gate_proj_weight/max_abs":0.001739501953125,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_69_post_attention_layernorm/norm":5792.5944824244525,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/max_abs":0.000896453857421875,"train/train/tensor_param_model_layers_81_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/std":0.0007490564781521543,"train/train/tensor_act_model_layers_85_mlp_gate_proj/max_abs":1.046875,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/max_abs":0.00421142578125,"train/train/tensor_act_model_layers_25_self_attn/mean":-0.001499176025390625,"train/train/tensor_act_model_layers_16_input_layernorm/norm":5792.525268555319,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/max_abs":0.00994873046875,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_35/act/max_abs":5.125,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/std":0.00037960693225853626,"train/train/layer_model_layers_30/act/mean":-0.0065174102783203125,"train/train/tensor_act_model_layers_78_self_attn_v_proj/max_abs":1.078125,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/mean":-9.255018085241318e-09,"train/train/tensor_act_model_layers_32_self_attn_k_proj/std":0.22021761633973613,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/norm":0.00011985389949323157,"train/train/tensor_act_model_layers_27_input_layernorm/std":1.0000011380755385,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_42_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/norm":0.0001955026533536869,"train/train/layer_model_layers_21/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/mean":-4.307366907596588e-09,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/norm":0.0006515040593416272,"train/train/layer__model_layers_29/param/max_abs":1,"train/train/tensor_act_model_layers_7_self_attn/mean":0.000713348388671875,"train/train/tensor_act_model_layers_56_self_attn_q_proj/mean":-0.0103759765625,"train/train/tensor_act_model_layers_1_self_attn_o_proj/norm":108.08046583816427,"train/train/layer__model_layers_48/param/mean":0.0015781420441387968,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/max_abs":0.00043487548828125,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/max_abs":0.000720977783203125,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp_gate_proj/norm":1854.5916435547356,"train/train/tensor_param_model_layers_39_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_93_input_layernorm/max_abs":4.59375,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp/norm":88.23609882394013,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/norm":3.59375,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/std":1.4498850202444376e-06,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/max_abs":0.00189971923828125,"train/train/tensor_act_model_layers_60_mlp_down_proj/max_abs":0.0849609375,"train/train/tensor_act_model_layers_9/norm":820.1364418915136,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/std":0.0198974609375,"train/train/layer_model_layers_8/act/max_abs":5.03125,"train/train/layer_model_layers_82/grad/max_abs":0.0038604736328125,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/mean":2.0682811737060547e-05,"train/train/tensor_act_model_layers_93_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/std":7.494017210881537e-05,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/norm":0.08099841267007657,"train/train/tensor_act_model_layers_25_mlp_down_proj/mean":0.0006303787231445312,"train/train/tensor_act_model_layers_68_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp_gate_proj/norm":1784.712429200748,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/max_abs":2.2649765014648438e-05,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/mean":-8.883653208613396e-07,"train/train/tensor_param_model_layers_89_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/max_abs":5.632638931274414e-06,"train/train/tensor_act_model_layers_86_self_attn_q_proj/norm":1309.6353426211394,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/mean":-7.404014468193054e-07,"train/train/tensor_param_model_layers_26_mlp_gate_proj_weight/std":0.0198974609375,"train/train/layer_model_layers_72/grad/mean":2.937943147213522e-08,"train/train/tensor_grad_model_layers_13_mlp_gate_proj_weight/std":0.00018140513481942742,"train/train/tensor_act_model_layers_71_post_attention_layernorm/mean":-0.0014438629150390625,"train/train/tensor_act_model_layers_26_self_attn/max_abs":0.2392578125,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/mean":-3.7962308852002025e-09,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/norm":0.00024890534263062754,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/std":0.00039040843017537283,"train/train/tensor_act_model_layers_4_post_attention_layernorm/norm":5792.143554693005,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/norm":0.13003324817093054,"train/train/tensor_act_model_layers_82_self_attn/norm":289.35159843022666,"train/train/tensor_act_model_layers_90_self_attn_v_proj/mean":0.009521484375,"train/train/tensor_grad_model_layers_30_mlp_gate_proj_weight/std":0.00011499079854298455,"train/train/tensor_grad_model_layers_14_mlp_gate_proj_weight/mean":1.3993121683597565e-07,"train/train/tensor_act_model_layers_58_self_attn_q_proj/max_abs":1.1171875,"train/train/tensor_param_model_layers_17_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/mean":-8.726492524147034e-07,"train/train/tensor_act_model_layers_23_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn_v_proj/mean":0.00453948974609375,"train/train/tensor_act_model_layers_10_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/std":7.912474463490751e-07,"train/train/tensor_act_model_layers_45_self_attn_q_proj/max_abs":1.0625,"train/train/tensor_act_model_layers_27_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/std":0.0009760160233948543,"train/train/tensor_param_model_layers_80_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_o_proj/mean":0.0004374384880065918,"train/train/layer_model_layers_64/grad/std":0.00019473961351717964,"train/train/tensor_act_model_layers_15_self_attn_o_proj/max_abs":0.2412109375,"train/train/tensor_act_model_layers_50_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/max_abs":0.07861328125,"train/train/tensor_act_model_layers_8_mlp/std":0.016358192547126074,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_mlp_gate_proj_weight/mean":-7.757917046546936e-07,"train/train/tensor_act_model_layers_34_mlp/mean":-0.0004787445068359375,"train/train/tensor_param_model_layers_82_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/mean":5.859328666701913e-08,"train/train/tensor_act_model_layers_24_mlp/mean":0.00026679039001464844,"train/train/layer_model_layers_45/act/norm":9109.104487603086,"train/train/tensor_param_model_layers_78_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_64_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40/mean":-0.002719879150390625,"train/train/tensor_act_model_layers_14_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/std":7.103875132223004e-05,"train/train/tensor_act_model_layers_36_mlp_gate_proj/norm":1891.454659510152,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_mlp_gate_proj_weight/mean":-3.528594970703125e-05,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_v_proj/max_abs":1.21875,"train/train/tensor_act_model_layers_35_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_14_self_attn_q_proj/std":0.22168694864472166,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_input_layernorm/mean":-0.05218505859375,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/std":0.00010769190121194774,"train/train/tensor_act_model_layers_54_self_attn_k_proj/norm":1270.0461978052192,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_gate_proj_weight/max_abs":0.083984375,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_gate_proj/mean":0.0012617111206054688,"train/train/tensor_act_model_layers_37_self_attn_v_proj/std":0.22705275934649763,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/max_abs":7.420778274536133e-06,"train/train/tensor_act_model_layers_18/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn/max_abs":0.24609375,"train/train/tensor_act_model_layers_25_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/norm":0.0010785478901932995,"train/train/tensor_act_model_layers_62_self_attn_o_proj/norm":294.34059773673096,"train/train/tensor_act_model_layers_82_mlp_down_proj/mean":6.339699029922485e-05,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_74/act/max_abs":4.65625,"train/train/layer_model_layers_12/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/max_abs":0.000240325927734375,"train/train/tensor_act_model_layers_3_post_attention_layernorm/mean":-0.028656005859375,"train/train/layer_model_layers_45/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/norm":0.16336930378423395,"train/train/tensor_act_model_layers_69_self_attn_q_proj/norm":1277.2499883424985,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_input_layernorm/mean":0.004016876220703125,"train/train/tensor_act_model_layers_53_post_attention_layernorm/mean":-0.0041599273681640625,"train/train/tensor_act_model_layers_10_mlp_up_proj/norm":1837.380318858181,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/mean":3.0994415283203125e-06,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_input_layernorm/max_abs":4.96875,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/norm":0.009102427232782003,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0/max_abs":0.2265625,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/std":7.917296350251155e-05,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/mean":-0.00010442733764648438,"train/train/tensor_act_model_layers_31_self_attn_v_proj/max_abs":1.203125,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/max_abs":0.005401611328125,"train/train/tensor_param_model_layers_13_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_o_proj/norm":282.9645421103391,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/norm":0.0011455149077173779,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn/mean":0.0009365081787109375,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/std":0.0006441447819675111,"train/train/tensor_act_model_layers_31_mlp_up_proj/std":0.22754096101513205,"train/train/tensor_grad_model_layers_8_mlp_gate_proj_weight/std":0.00026631293208128866,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_gate_proj/std":0.22973743133715885,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/norm":0.028522493240746176,"train/train/tensor_act_model_layers_68_input_layernorm/std":1.0000037901395242,"train/train/tensor_grad_model_layers_71_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp_gate_proj/std":0.2265645957872452,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/norm":0.026248807425883713,"train/train/tensor_act_model_layers_23_post_attention_layernorm/norm":5792.554443359792,"train/train/tensor_act_model_layers_7_self_attn_k_proj/norm":1330.8174884175705,"train/train/tensor_act_model_layers_40_mlp_gate_proj/mean":0.014678955078125,"train/train/tensor_act_model_layers_91_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_mlp/mean":0.00035858154296875,"train/train/tensor_param_model_layers_93_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_v_proj/std":0.2309578828749949,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/max_abs":0.0308837890625,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/mean":-6.196205504238605e-08,"train/train/layer__model_layers_49/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62/mean":0.0015439987182617188,"train/train/tensor_act_model_layers_21_mlp_gate_proj/max_abs":1.171875,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_14/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/max_abs":0.0003757476806640625,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/max_abs":0.001190185546875,"train/train/layer__model_layers_71/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_input_layernorm/mean":0.0016231536865234375,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/norm":0.0034356511356653198,"train/train/layer__model_layers_28/param/std":0.04427693624227387,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/max_abs":0.003997802734375,"train/train/layer__model_layers_32/param/std":0.0442331253586111,"train/train/tensor_act_model_layers_4_mlp/mean":-0.0007181167602539062,"train/train/tensor_act_model_layers_14_self_attn_v_proj/mean":-0.004520416259765625,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/norm":0.045459679291983086,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_gate_proj_weight/max_abs":0.0052490234375,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/std":0.019775390625,"train/train/tensor_param_model_layers_43_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_1_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/std":7.849713768853493e-05,"train/train/tensor_param_model_layers_29_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/norm":0.00014509724577094904,"train/train/tensor_act_model_layers_75_self_attn_q_proj/max_abs":1.0234375,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/mean":3.5552147892303765e-09,"train/train/tensor_act_model_layers_21_post_attention_layernorm/mean":-0.0570068359375,"train/train/tensor_act_model_layers_38_self_attn_o_proj/norm":295.6614942950424,"train/train/tensor_act_model_layers_71_input_layernorm/mean":-0.0026874542236328125,"train/train/tensor_act_model_layers_46_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_40_self_attn_k_proj/max_abs":1.1796875,"train/train/tensor_act_model_layers_29_self_attn/std":0.047792070188501366,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/norm":0.0009184987802969333,"train/train/tensor_act_model_layers_24_post_attention_layernorm/norm":5792.561035161714,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_o_proj/mean":-0.0001914501190185547,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/mean":-8.003553375601768e-10,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_73/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_67_mlp/mean":0.0004515647888183594,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/max_abs":0.0078125,"train/train/tensor_act_model_layers_35_mlp_gate_proj/max_abs":1.109375,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/std":5.172158627014648e-07,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/mean":4.446519596967846e-09,"train/train/tensor_act_model_layers_18_post_attention_layernorm/std":1.0000040288933418,"train/train/layer_model_layers_27/act/norm":9015.882897696441,"train/train/tensor_act_model_layers_47_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_mlp_gate_proj_weight/mean":-1.3113021850585938e-05,"train/train/tensor_act_model_layers_25_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn_v_proj/std":0.22193285375020358,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_42_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/max_abs":0.00020599365234375,"train/train/tensor_act_model_layers_41_post_attention_layernorm/norm":5792.580688476793,"train/train/layer__model_layers_23/param/std":0.04422441635505548,"train/train/tensor_param_model_layers_80_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_27_mlp/mean":-0.0002856254577636719,"train/train/tensor_param_model_layers_28_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_gate_proj/std":0.2260753366883487,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_88_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/norm":0.05322671577300525,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/max_abs":0.0009002685546875,"train/train/layer_model_layers_5/act/mean":-0.008914879390171595,"train/train/tensor_grad_model_layers_14_mlp_gate_proj_weight/norm":0.06688250134189279,"train/train/tensor_act_model_layers_55_self_attn_o_proj/mean":0.00164794921875,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/max_abs":0.08349609375,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/max_abs":0.0869140625,"train/train/tensor_param_model_layers_10_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_input_layernorm/mean":-0.04693603515625,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/mean":8.58306884765625e-05,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_mlp_gate_proj_weight/norm":0.03938598207183772,"train/train/layer_model_layers_28/grad/norm":0.24412849684529098,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn_o_proj/norm":277.4245531201896,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_56/act/std":0.4218408092617341,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/mean":-3.62437276635319e-07,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_q_proj/norm":1301.7907588378118,"train/train/tensor_act_model_layers_77_mlp_down_proj/norm":92.35353431537136,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_o_proj/norm":280.3868408330669,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/mean":0.00017261505126953125,"train/train/tensor_act_model_layers_52_self_attn_v_proj/max_abs":1.078125,"train/train/layer__model_layers_4/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_62/grad/mean":-1.0850593591524453e-07,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/std":2.3918759095493353e-05,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/max_abs":0.0791015625,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/mean":-0.00011539459228515625,"train/train/tensor_act_model_layers_30_post_attention_layernorm/mean":-0.03814697265625,"train/train/tensor_param_model_layers_84_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_59/act/mean":-0.002478889056614467,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp_gate_proj/mean":0.00024580955505371094,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_mlp/std":0.015671022464751654,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/max_abs":0.0859375,"train/train/tensor_act_model_layers_74_self_attn_o_proj/std":0.04980673846918754,"train/train/tensor_act_model_layers_47_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_q_proj/max_abs":1.0546875,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/std":3.6296802446967025e-07,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/mean":2.4847686290740967e-06,"train/train/tensor_act_model_layers_3_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_73_self_attn_o_proj/max_abs":0.244140625,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/mean":-9.632110595703125e-05,"train/train/tensor_grad_model_layers_19_mlp_gate_proj_weight/norm":0.05813405591908632,"train/train/tensor_act_model_layers_24_mlp_gate_proj/mean":0.003387451171875,"train/train/tensor_param_model_layers_91_mlp_gate_proj_weight/std":0.0201416015625,"train/train/layer_model_layers_29/act/std":0.4167962813265762,"train/train/tensor_act_model_layers_43_self_attn_o_proj/norm":276.3456489040801,"train/train/tensor_act_model_layers_46_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_14/grad/mean":1.3813341738199946e-06,"train/train/tensor_act_model_layers_42_self_attn_q_proj/max_abs":1.03125,"train/train/tensor_act_model_layers_73_input_layernorm/norm":5792.596679688806,"train/train/tensor_act_model_layers_24_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/std":0.00015678118409884908,"train/train/layer__model_layers_40/param/std":0.044253647067485094,"train/train/tensor_param_model_layers_71_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp/norm":92.04833459688456,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/mean":0.00010967254638671875,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_22_mlp_down_proj/mean":-0.0005245208740234375,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_12/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/std":6.719566100533252e-07,"train/train/tensor_act_model_layers_34_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/mean":-1.609325408935547e-05,"train/train/tensor_act_model_layers_6_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_q_proj/std":0.21972783565152393,"train/train/tensor_act_model_layers_47_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/std":1.1952306485623539e-06,"train/train/tensor_act_model_layers_29_self_attn_q_proj/max_abs":1.09375,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/mean":-1.6823410987854004e-05,"train/train/tensor_act_model_layers_87_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_32_self_attn_k_proj/mean":-0.00818634033203125,"train/train/tensor_act_model_layers_14_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/mean":-4.601478576660156e-05,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/mean":-0.0002899169921875,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp_down_proj/std":0.015366008009653922,"train/train/layer_model_layers_25/grad/norm":0.25621645789699493,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/max_abs":0.08154296875,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn/norm":296.95200205864137,"train/train/tensor_act_model_layers_90_post_attention_layernorm/mean":0.0122222900390625,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_24_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/norm":3.65625,"train/train/tensor_param_model_layers_20_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn/norm":272.3716872119266,"train/train/tensor_act_model_layers_67_mlp_up_proj/std":0.22680730714221406,"train/train/layer__model_layers_71/param/norm":17.937302644820235,"train/train/tensor_act_model_layers_57_mlp_gate_proj/max_abs":1.1484375,"train/train/tensor_act_model_layers_91_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/max_abs":0.004852294921875,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/norm":0.0028256274036959623,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/max_abs":0.007659912109375,"train/train/tensor_act_model_layers_61_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_post_attention_layernorm/mean":0.000125885009765625,"train/train/tensor_act_model_layers_83_post_attention_layernorm/mean":0.0056915283203125,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/max_abs":0.002105712890625,"train/train/tensor_act_model_layers_44_self_attn_k_proj/max_abs":1.0625,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/mean":0.00010251998901367188,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/max_abs":0.0012969970703125,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/max_abs":0.00115966796875,"train/train/tensor_act_model_layers_85_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_self_attn_k_proj/mean":0.00464630126953125,"train/train/tensor_act_model_layers_45_mlp_up_proj/mean":-0.007598876953125,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/std":0.00043287262155407416,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/max_abs":0.00848388671875,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_v_proj/norm":1338.5513235902376,"train/train/tensor_param_model_layers_64_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_18_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_12/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/std":8.946847499586781e-05,"train/train/tensor_act_model_layers_53_mlp_up_proj/std":0.22534311760156833,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_14_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_input_layernorm/mean":-0.0025463104248046875,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_33_mlp_gate_proj_weight/std":0.00010912542871874555,"train/train/tensor_act_model_layers_78_input_layernorm/max_abs":4.65625,"train/train/tensor_act_model_layers_11_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_gate_proj/max_abs":1.2578125,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_68/act/norm":9227.5808127457,"train/train/layer__model_layers_42/param/max_abs":1,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/norm":0.02685976860996258,"train/train/tensor_act_model_layers_86_mlp/norm":90.91961238020785,"train/train/tensor_param_model_layers_69_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/max_abs":0.09033203125,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_12/max_abs":0.83984375,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_65/grad/mean":3.881633714678321e-08,"train/train/tensor_act_model_layers_19_input_layernorm/norm":5792.542846680493,"train/train/tensor_act_model_layers_72_mlp/max_abs":0.07958984375,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp_gate_proj/mean":-0.0020318031311035156,"train/train/tensor_act_model_layers_85_mlp/norm":92.75920276238986,"train/train/tensor_param_model_layers_67_input_layernorm_weight/max_abs":1,"train/train/layer__model_layers_40/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/norm":0.0009021275126280384,"train/train/tensor_act_model_layers_78_self_attn_q_proj/mean":-0.0024051666259765625,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/max_abs":1.2874603271484375e-05,"train/train/tensor_act_model_layers_46_mlp_down_proj/norm":89.25947714919363,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/max_abs":0.08642578125,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_9_mlp_down_proj/std":0.015259784904408845,"train/train/tensor_act_model_layers_81/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/act/std":0.4170791733784959,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_77/param/mean":0.0017020453156993468,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/max_abs":3.993511199951172e-06,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/max_abs":0.08251953125,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/std":5.34106380744129e-07,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/mean":-2.1219253540039062e-05,"train/train/layer__model_layers_91/param/max_abs":1,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_k_proj/std":0.22290634239247809,"train/train/tensor_act_model_layers_93_input_layernorm/std":1.000003227027514,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_49/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78/max_abs":2.0625,"train/train/tensor_param_model_layers_55_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_72_self_attn_v_proj/norm":1308.465287850587,"train/train/layer_model_layers_79/grad/norm":0.1588906540685139,"train/train/tensor_act_model_layers_16_self_attn_o_proj/norm":295.3285272532396,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/max_abs":0.00421142578125,"train/train/tensor_act_model_layers_14_self_attn_v_proj/max_abs":1.09375,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/norm":0.7732377227785349,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/mean":-9.741634130477905e-07,"train/train/tensor_act_model_layers_24_self_attn_v_proj/mean":0.000614166259765625,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/norm":0.10535283253052379,"train/train/tensor_param_model_layers_69_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/mean":-0.0002460479736328125,"train/train/tensor_param_model_layers_22_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_post_attention_layernorm/mean":-0.00313568115234375,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/mean":-0.00013828277587890625,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_41/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_post_attention_layernorm/norm":5791.870971685228,"train/train/tensor_act_model_layers_63_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_5_mlp_down_proj/std":0.01562518882451105,"train/train/tensor_act_model_layers_15_post_attention_layernorm/max_abs":4.8125,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/mean":-0.000148773193359375,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/std":4.456215540622299e-07,"train/train/tensor_grad_model_layers_48_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_24_mlp_gate_proj_weight/mean":-0.00020313262939453125,"train/train/tensor_act_model_layers_13/std":0.17187998165692267,"train/train/tensor_act_model_layers_14_input_layernorm/norm":5792.514892578398,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/mean":1.7642974853515625e-05,"train/train/tensor_act_model_layers_36_mlp_down_proj/max_abs":0.09228515625,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/max_abs":0.07666015625,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/mean":-0.00014495849609375,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/mean":-6.771087646484375e-05,"train/train/tensor_act_model_layers_12/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_18_mlp_gate_proj_weight/norm":0.06082005554027336,"train/train/tensor_act_model_layers_68_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_41_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_embed_tokens/std":0.020263673442048356,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/max_abs":0.0849609375,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/mean":-2.291053533554077e-06,"train/train/tensor_param_model_layers_51_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/mean":-0.00021076202392578125,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/mean":-1.0634266800479963e-09,"train/train/tensor_act_model_layers_57_self_attn_k_proj/norm":1323.4586509277722,"train/train/tensor_act_model_layers_64_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/norm":0.0003640238701628173,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_input_layernorm/std":1.0000085969635335,"train/train/tensor_act_model_layers_20_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_77/grad/norm":0.15735316429412124,"train/train/tensor_act_model_layers_23_mlp_up_proj/norm":1876.6317119709088,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_77_mlp_gate_proj/norm":1859.0947941507031,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/max_abs":0.09375,"train/train/tensor_act_model_layers_23_self_attn_o_proj/norm":282.9412194841048,"train/train/tensor_act_model_layers_24_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_53/param/std":0.04421860082847485,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/max_abs":0.006011962890625,"train/train/layer_model_layers_77/grad/mean":8.916233153359939e-08,"train/train/tensor_act_model_layers_80_input_layernorm/norm":5792.595581055289,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_37_mlp_gate_proj/std":0.2263221489536017,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/max_abs":0.08642578125,"train/train/tensor_grad_model_layers_72_mlp_gate_proj_weight/max_abs":0.000926971435546875,"train/train/layer_model_layers_87/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_88_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/std":0.0005250513574003201,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/std":4.0823015456081424e-07,"train/train/tensor_act_model_layers_32_mlp_gate_proj/std":0.22339173403449494,"train/train/tensor_act_model_layers_4_post_attention_layernorm/mean":-0.046630859375,"train/train/tensor_act_model_layers_1/std":0.036255324927758625,"train/train/layer_model_layers_0/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/mean":9.052455425262451e-06,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/max_abs":0.08642578125,"train/train/layer_model_layers_52/act/max_abs":4.84375,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/norm":0.0008025587267090058,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/max_abs":0.09130859375,"train/train/tensor_act_model_layers_60_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_68_mlp/std":0.01599240766133549,"train/train/layer__model_layers_75/param/mean":0.0014225250101312645,"train/train/tensor_act_model_layers_84_mlp_down_proj/std":0.015518404410028785,"train/train/tensor_act_model_layers_77_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/std":7.08629621355674e-05,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/mean":-1.3597309589385986e-06,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/mean":-0.00020503997802734375,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_70/act/mean":-0.0016814566084316798,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/mean":1.1129304766654968e-06,"train/train/tensor_grad_model_layers_9_mlp_gate_proj_weight/norm":0.07574342393185629,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_78_post_attention_layernorm/mean":0.0019931793212890625,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/max_abs":0.0003814697265625,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/max_abs":0.00070953369140625,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_32_mlp_up_proj/max_abs":1.1640625,"train/train/tensor_param_model_layers_3_mlp_gate_proj_weight/max_abs":0.080078125,"train/train/tensor_act_model_layers_33_input_layernorm/std":1.0000012181691857,"train/train/layer_model_layers_47/act/max_abs":4.78125,"train/train/layer__model_layers_62/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp_down_proj/mean":-0.0003205537796020508,"train/train/layer__model_layers_21/param/norm":17.92654690631676,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/max_abs":0.08837890625,"_runtime":85,"train/train/tensor_act_model_layers_84_mlp_up_proj/max_abs":1.1640625,"train/train/tensor_grad_model_layers_71_mlp_gate_proj_weight/norm":0.02741817155918712,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/max_abs":0.00714111328125,"train/train/tensor_act_model_layers_30_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/std":3.5123908934234683e-07,"train/train/tensor_act_model_layers_82_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_74_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/mean":-0.00022029876708984375,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/std":4.5992400940632455e-07,"train/train/tensor_act_model_layers_54_mlp_gate_proj/std":0.22852040694910586,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/std":0.00024625122618852804,"train/train/tensor_act_model_layers_32/mean":-0.00684356689453125,"train/train/tensor_act_model_layers_76_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_41/act/norm":9092.738442349182,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_79_self_attn_o_proj/std":0.05237228230668631,"train/train/tensor_act_model_layers_45_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_mlp_gate_proj_weight/std":0.00010293308467868134,"train/train/tensor_act_model_layers_44_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_up_proj/norm":1840.6161189046904,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_54_self_attn_q_proj/norm":1281.057435563393,"train/train/tensor_act_model_layers_42_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/max_abs":0.00099945068359375,"train/train/layer_model_layers_8/act/mean":-0.012107329709189278,"train/train/tensor_act_model_layers_61_mlp_up_proj/std":0.2243679774690542,"train/train/layer__model_layers_33/param/norm":17.9296857979302,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_lm_head/max_abs":1.2734375,"train/train/tensor_act_model_layers_23_self_attn/mean":0.00222015380859375,"train/train/tensor_act_model_layers_12_self_attn_o_proj/max_abs":0.2392578125,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_input_layernorm/max_abs":4.71875,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/max_abs":5.21540641784668e-06,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_54/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/mean":9.387731552124023e-07,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/norm":0.044220802609955284,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/norm":0.09066069327573453,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_78/param/max_abs":1,"train/train/tensor_act_model_layers_86_self_attn_k_proj/norm":1300.2840219250718,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/mean":-5.6743621826171875e-05,"train/train/layer__model_layers_93/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_46_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/norm":0.024483239396471,"train/train/tensor_act_model_layers_69_mlp/norm":89.58783643347356,"train/train/tensor_act_model_layers_59_self_attn/norm":287.5581165163929,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/norm":0.0008271188261987535,"train/train/tensor_act_model_layers_76_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_70/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/max_abs":0.03173828125,"train/train/tensor_act_model_layers_16_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/mean":3.918103175237775e-09,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/norm":0.00029643628082709196,"train/train/tensor_act_model_layers_76_self_attn_v_proj/std":0.23023511067749577,"train/train/tensor_act_model_layers_64_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57/frac_near_dtype_limit":0,"train/train/layer_model_layers_80/grad/mean":1.2099547471484975e-07,"train/train/tensor_grad_model_layers_27_mlp_gate_proj_weight/std":0.00012531052028065813,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/norm":0.03706135905987927,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/mean":7.182825356721878e-08,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/std":8.441298834237592e-05,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/norm":0.030283772334974184,"train/train/tensor_param_model_layers_31_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_16_input_layernorm/mean":-0.06268310546875,"train/train/tensor_act_model_layers_30/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_30/norm":1556.1592411733523,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/max_abs":0.0751953125,"train/train/tensor_act_model_layers_7/max_abs":0.6796875,"train/train/tensor_act_model_layers_25_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/norm":0.26983227712586694,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_43/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_69/param/norm":17.936363476625495,"train/train/tensor_act_model_layers_56_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/mean":-2.8442591428756714e-06,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_35_input_layernorm/mean":-0.020263671875,"train/train/tensor_act_model_layers_72_self_attn_k_proj/max_abs":1.109375,"train/train/tensor_act_model_layers_59_self_attn_v_proj/mean":0.00360870361328125,"train/train/tensor_param_model_layers_26_mlp_gate_proj_weight/mean":-0.00018405914306640625,"train/train/tensor_act_model_layers_55_self_attn_q_proj/max_abs":1.078125,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/max_abs":0.0859375,"train/train/tensor_act_model_layers_7_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/norm":0.00021630229124064743,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/mean":-1.601874828338623e-06,"train/train/tensor_act_model_layers_46_mlp_gate_proj/norm":1870.2694763207405,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/mean":2.992237568832934e-09,"train/train/tensor_act_model_layers_93_self_attn/max_abs":0.23046875,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn/max_abs":0.2265625,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_66/norm":2353.350836765558,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_56_mlp_up_proj/max_abs":1.0625,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/max_abs":0.0031890869140625,"train/train/tensor_param_model_layers_51_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/mean":2.6158522814512253e-07,"train/train/tensor_act_model_layers_36_mlp/std":0.01617447514501086,"train/train/tensor_grad_model_layers_72_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71/std":0.4209092247805945,"train/train/tensor_act_model_layers_31_mlp_gate_proj/mean":0.009674072265625,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/max_abs":0.1005859375,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/std":0.02001953125,"train/train/layer_model_layers_82/act/max_abs":4.78125,"train/train/layer_model_layers_43/act/std":0.41975525997186863,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_20_mlp/max_abs":0.08935546875,"train/train/layer__model_layers_83/param/mean":0.0016399746566034517,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/max_abs":0.08251953125,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_14_mlp_gate_proj/std":0.22851652444091639,"train/train/tensor_act_model_layers_33_self_attn_o_proj/mean":0.0007238388061523438,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_k_proj/norm":1312.5264552521667,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/std":0.0007471798841533486,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/max_abs":3.6507844924926758e-06,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/max_abs":0.001129150390625,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/max_abs":0.00107574462890625,"train/train/tensor_act_model_layers_27/mean":-0.0085296630859375,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_9_mlp_gate_proj_weight/mean":-4.1961669921875e-05,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_41/grad/std":0.0002358230157322759,"train/train/tensor_act_model_layers_87_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/mean":4.755565896630287e-08,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/mean":0.000186920166015625,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/std":3.588925492812873e-07,"train/train/tensor_act_model_layers_10_mlp/std":0.015549359624922337,"train/train/layer_model_layers_68/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn_o_proj/mean":0.0007529258728027344,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/norm":0.00011753358879474592,"train/train/tensor_act_model_layers_14_self_attn_o_proj/std":0.048890890866537265,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/max_abs":0.0859375,"train/train/tensor_act_model_layers_59/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_gate_proj/std":0.2277885273191421,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/mean":1.7280399333685637e-09,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/mean":-6.103515625e-05,"train/train/tensor_act_model_layers_6_self_attn_v_proj/max_abs":1.34375,"train/train/tensor_grad_model_layers_8_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_38_self_attn_q_proj/std":0.22607864451247356,"train/train/tensor_act_model_layers_36_post_attention_layernorm/norm":5792.570556642194,"train/train/layer_model_layers_58/act/max_abs":4.6875,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55/max_abs":1.8046875,"train/train/tensor_act_model_layers_76_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/max_abs":0.0810546875,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_71_mlp_up_proj/norm":1876.749135056742,"train/train/tensor_act_model_layers_89_self_attn/max_abs":0.2197265625,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/mean":4.649162292480469e-05,"train/train/layer__model_layers_70/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/max_abs":0.00238037109375,"train/train/tensor_act_model_layers_42_post_attention_layernorm/max_abs":4.875,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/mean":1.907028490677476e-08,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/max_abs":0.0017242431640625,"train/train/tensor_act_model_layers_46_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/mean":6.063783075660467e-08,"train/train/tensor_act_model_layers_24_self_attn_k_proj/std":0.22339604589011836,"train/train/tensor_grad_model_layers_62_mlp_gate_proj_weight/norm":0.028050618360200782,"train/train/tensor_act_model_layers_19_self_attn_v_proj/max_abs":1.1796875,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/mean":1.755543053150177e-06,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/mean":-3.5408884286880493e-06,"train/train/tensor_act_model_layers_12_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/std":0.0001009110725359331,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/max_abs":0.0009002685546875,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/max_abs":0.09375,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/max_abs":0.09130859375,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/mean":1.505486579844728e-08,"train/train/tensor_act_model_layers_38_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_q_proj/norm":1343.2072617645524,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp_down_proj/mean":-6.3478946685791016e-06,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/norm":0.1780975320618275,"train/train/tensor_act_model_layers_40_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_gate_proj/mean":-0.004413604736328125,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/mean":-6.124377250671387e-06,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_input_layernorm/norm":5792.551635744825,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/max_abs":0.000186920166015625,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/norm":0.02844997844862834,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/mean":2.7881469577550888e-08,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/norm":0.000726509569948601,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_post_attention_layernorm/norm":5792.580078125895,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/norm":0.08640819305939285,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/max_abs":4.0531158447265625e-06,"train/train/tensor_act_model_layers_27_mlp_gate_proj/norm":1852.3976801008341,"train/train/tensor_act_model_layers_41_self_attn/mean":-0.00014269351959228516,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_mlp_up_proj/std":0.22705251282241895,"train/train/tensor_act_model_layers_33_self_attn_k_proj/mean":0.00496673583984375,"train/train/tensor_act_model_layers_50_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/norm":0.005440352141293788,"train/train/tensor_act_model_layers_86/max_abs":2.140625,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/std":0.0001257731590332944,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/norm":2.59375,"train/train/tensor_act_model_layers_63_mlp_down_proj/std":0.01577854052953808,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_gate_proj_weight/max_abs":0.0869140625,"train/train/layer_model_layers_55/grad/max_abs":0.005584716796875,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/max_abs":0.00011348724365234375,"train/train/tensor_act_model_layers_17/max_abs":0.97265625,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/max_abs":0.0927734375,"train/train/layer_model_layers_40/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn_q_proj/mean":0.013397216796875,"train/train/layer__model_layers_24/param/norm":17.933266595387746,"train/train/tensor_param_model_layers_17_mlp_gate_proj_weight/mean":-0.00014591217041015625,"train/train/layer_model_layers_91/act/max_abs":4.5625,"train/train/tensor_act_model_layers_15_input_layernorm/max_abs":5.03125,"train/train/tensor_act_model_layers_60/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/max_abs":0.0034332275390625,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_64/param/max_abs":1,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/norm":2.53125,"train/train/tensor_act_model_layers_87_self_attn/mean":-0.0015964508056640625,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/max_abs":0.00015354156494140625,"train/train/tensor_act_model_layers_57_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn_q_proj/max_abs":1.0546875,"train/train/layer_model_layers_81/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp/norm":86.55243558355863,"train/train/tensor_act_model_layers_40_self_attn_o_proj/max_abs":0.2265625,"train/train/tensor_param_model_layers_9_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/std":0.0007563816607769195,"train/train/layer_model_layers_65/act/norm":9214.693021933757,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/mean":2.1364539861679077e-06,"train/train/layer_model_layers_57/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9/mean":-0.00740814208984375,"train/train/layer_model_layers_64/grad/mean":6.95592444825935e-08,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7/norm":685.3050313294224,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_90/grad/mean":8.199499114002457e-08,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/std":6.66484681649684e-05,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/max_abs":0.00118255615234375,"train/train/layer_model_layers_66/grad/norm":0.16497527807572482,"train/train/tensor_act_model_layers_49_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/mean":-0.00019741058349609375,"train/train/layer_model_layers_20/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/max_abs":0.080078125,"train/train/tensor_act_model_layers_57_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/std":1.315572264371641e-06,"train/train/layer_model_layers_81/grad/std":0.00018247856324744814,"train/train/tensor_act_model_layers_74_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_8/act/norm":8940.018612079153,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/max_abs":0.083984375,"train/train/layer__model_layers_2/param/std":0.04428045771379499,"train/train/tensor_act_model_layers_40_self_attn_v_proj/mean":0.00862884521484375,"train/train/tensor_param_model_layers_77_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_60_mlp_gate_proj_weight/mean":7.906928658485413e-07,"train/train/tensor_param_model_layers_34_mlp_gate_proj_weight/mean":7.343292236328125e-05,"train/train/layer_model_layers_77/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/max_abs":0.0021820068359375,"train/train/tensor_act_model_layers_48_self_attn_o_proj/norm":291.21136793218284,"train/train/tensor_act_model_layers_43_mlp_gate_proj/std":0.22632031747877268,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/mean":-5.587935447692871e-06,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/std":5.304554709016439e-07,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/std":0.02001953125,"train/train/layer__model_layers_0/param/std":0.04423906603202986,"train/train/tensor_act_model_layers_42/norm":1837.821848428952,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_self_attn_o_proj/max_abs":0.2275390625,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/norm":0.02472413567902052,"train/train/tensor_act_model_layers_25_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_mlp_up_proj/norm":1896.2007100405128,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_52_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn/mean":-0.0010232925415039062,"train/train/tensor_act_model_layers_41_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50/std":0.3496190171670284,"train/train/layer__model_layers_37/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/mean":-0.000316619873046875,"train/train/tensor_act_model_layers_52_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/mean":-7.830560207366943e-06,"train/train/layer_model_layers_3/act/max_abs":5.03125,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/std":8.213617266503985e-05,"train/train/tensor_act_model_layers_62_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_lm_head/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp_up_proj/norm":1844.2759295645528,"train/train/tensor_act_model_layers_4_self_attn_o_proj/norm":254.09113697599594,"train/train/layer_model_layers_47/act/norm":9134.461663516038,"train/train/tensor_act_model_layers_74/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_11/std":0.15625236256124586,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/max_abs":4.351139068603516e-06,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_v_proj/max_abs":1.0390625,"train/train/layer__model_layers_39/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84/mean":0.00542449951171875,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/std":5.99379711942985e-07,"train/train/tensor_act_model_norm/mean":0.01849365234375,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_53/act/max_abs":4.9375,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/std":0.0030204631451768256,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/max_abs":0.09423828125,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/std":4.765754555941542e-05,"train/train/tensor_act_model_layers_34/std":0.28808816084569544,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/max_abs":0.004364013671875,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/mean":-2.034008502960205e-06,"train/train/tensor_act_model_layers_42_mlp/mean":-0.0003559589385986328,"train/train/tensor_act_model_layers_22_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/norm":0.14252209542283548,"train/train/tensor_param_model_layers_78_mlp_gate_proj_weight/mean":7.43865966796875e-05,"train/train/layer__model_layers_27/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/max_abs":0.0791015625,"train/train/tensor_act_model_layers_40_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_mlp_down_proj/max_abs":0.0869140625,"train/train/tensor_act_model_layers_73_self_attn_k_proj/max_abs":1.125,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/max_abs":0.005340576171875,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/norm":0.0007019742640577214,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_2_self_attn_v_proj/norm":1313.8771824186267,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/std":0.019775390625,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/std":7.636903862760391e-07,"train/train/layer_model_layers_43/grad/norm":0.18446888543978487,"train/train/tensor_param_model_layers_9_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/std":9.636068768729462e-05,"train/train/layer__model_layers_33/param/max_abs":1,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_15_mlp_gate_proj_weight/std":0.0001712271709572347,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/max_abs":0.0004863739013671875,"train/train/tensor_act_model_layers_73_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_72/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/norm":0.1596478090770908,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/mean":-1.1012889444828033e-06,"train/train/tensor_act_model_layers_85_self_attn/norm":291.14040547445387,"train/train/layer__model_layers_34/param/max_abs":1,"train/train/tensor_act_model_layers_7_self_attn_o_proj/max_abs":0.2236328125,"train/train/layer_model_layers_81/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_gate_proj/mean":-0.00021755695343017578,"train/train/layer_model_layers_91/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn/mean":-0.0006821155548095703,"train/train/tensor_act_model_layers_62_mlp/max_abs":0.07763671875,"train/train/tensor_act_model_layers_15_mlp_gate_proj/max_abs":1.1484375,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/norm":0.31406772286546486,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/mean":-0.000194549560546875,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/mean":-0.00016498565673828125,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/max_abs":7.12275505065918e-06,"train/train/tensor_act_model_layers_81_mlp/max_abs":0.080078125,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_param_model_embed_tokens_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_mlp_up_proj/std":0.22461242139217155,"train/train/layer__model_layers_80/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_input_layernorm/norm":5792.5833740237495,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/std":0.019775390625,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/mean":8.843839168548584e-06,"train/train/tensor_act_model_layers_83_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/std":0.0011166719356383197,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/mean":-4.0838494896888733e-07,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/norm":0.005786502198608987,"train/train/tensor_act_model_layers_9_self_attn_v_proj/norm":1318.6141665212742,"train/train/tensor_param_model_layers_73_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/mean":0.00011444091796875,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/max_abs":0.09130859375,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/std":3.6593791078270084e-07,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/max_abs":0.08154296875,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_mlp_gate_proj_weight/max_abs":0.000919342041015625,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_post_attention_layernorm/max_abs":5.40625,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/max_abs":0.0849609375,"train/train/tensor_act_model_layers_50_self_attn_q_proj/max_abs":1.03125,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/max_abs":0.09619140625,"train/train/tensor_param_model_layers_88_mlp_gate_proj_weight/mean":-1.055002212524414e-05,"train/train/tensor_act_model_layers_32_self_attn_o_proj/mean":0.0011615753173828125,"train/train/tensor_act_model_layers_25_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_post_attention_layernorm/mean":-0.0038933753967285156,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/std":0.02001953125,"train/train/layer__model_layers_85/param/mean":0.0016494846195214995,"train/train/tensor_grad_model_layers_81_mlp_gate_proj_weight/max_abs":0.00102996826171875,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/max_abs":0.00084686279296875,"train/train/tensor_act_model_layers_21_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_61/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_mlp_gate_proj_weight/std":6.918912521801496e-05,"train/train/tensor_param_model_layers_83_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_v_proj/norm":1269.3871977932308,"train/train/tensor_act_model_layers_2_mlp_up_proj/norm":1855.5004284916604,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/std":8.503895772848197e-07,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/norm":0.024914159028759552,"train/train/tensor_act_model_layers_51_mlp_up_proj/max_abs":1.140625,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/max_abs":0.0004100799560546875,"train/train/tensor_act_model_layers_91/norm":2800.0263771984473,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_mlp/std":0.015228798331347491,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/norm":0.0015532611607105397,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/mean":-0.00017070770263671875,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/max_abs":0.0859375,"train/train/tensor_param_model_layers_48_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp_gate_proj/max_abs":1.15625,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_58_mlp_gate_proj_weight/max_abs":0.00118255615234375,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn_q_proj/mean":-0.0108184814453125,"train/train/tensor_param_model_layers_30_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_15/grad/frac_near_user_limit":0,"train/train/layer__model_layers_26/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_3_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/mean":-3.24249267578125e-05,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/std":7.576769095127949e-05,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_3_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_input_layernorm/max_abs":4.84375,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/mean":1.545250415802002e-05,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/mean":5.221366882324219e-05,"train/train/tensor_act_model_layers_53_self_attn/max_abs":0.2177734375,"train/train/tensor_act_model_layers_2/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/max_abs":0.00077056884765625,"train/train/tensor_act_model_layers_16_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/std":3.9654423514025043e-07,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/norm":0.23302068426962735,"train/train/layer_model_layers_71/grad/max_abs":0.005340576171875,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_q_proj/norm":1289.0662525416762,"train/train/tensor_param_model_layers_50_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_2_mlp/mean":-0.0002238750457763672,"train/train/tensor_act_model_layers_8_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_q_proj/std":0.22656417852609279,"train/train/tensor_act_model_layers_77_self_attn_o_proj/norm":289.9886559105947,"train/train/tensor_act_model_layers_56_self_attn_q_proj/max_abs":1.1015625,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/std":6.664031187032952e-07,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/mean":-7.963180541992188e-05,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/mean":1.5553086996078491e-06,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/norm":0.16854617206727146,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/max_abs":0.0037689208984375,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_16_self_attn_o_proj/std":0.05102959267746651,"train/train/tensor_act_model_layers_30_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/std":4.3841363424293696e-07,"train/train/tensor_act_model_layers_57_mlp/max_abs":0.07958984375,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/mean":1.6405247151851654e-06,"train/train/tensor_act_model_layers_53_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/norm":0.0028558913940932643,"train/train/tensor_act_model_layers_87_mlp_up_proj/norm":1865.7715678346024,"train/train/tensor_act_model_layers_18_mlp_down_proj/std":0.015717023452114582,"train/train/tensor_act_model_layers_27_mlp_down_proj/norm":86.55243558355863,"train/train/tensor_act_model_layers_31_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn/max_abs":0.2080078125,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/mean":1.0952353477478027e-05,"train/train/tensor_act_model_layers_60_self_attn/mean":-0.0014600753784179688,"train/train/tensor_act_model_layers_19_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/grad/std":0.0005134286150783229,"train/train/tensor_act_model_layers_29_input_layernorm/norm":5792.5733642619425,"train/train/tensor_param_model_layers_15_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_8_mlp_gate_proj/norm":1841.4734767509945,"train/train/tensor_param_model_layers_90_mlp_gate_proj_weight/max_abs":0.08642578125,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_k_proj/std":0.22192698970594776,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/mean":-0.00012159347534179688,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/std":0.00013246404800502232,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_81_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/std":6.315009156117901e-05,"train/train/tensor_act_model_layers_29_self_attn_o_proj/mean":-0.00130462646484375,"train/train/tensor_act_model_layers_38_mlp/norm":89.83683625483374,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/mean":0.00016117095947265625,"train/train/tensor_act_model_layers_0_mlp_down_proj/max_abs":0.0869140625,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_32/act/norm":9030.720334174615,"train/train/tensor_grad_model_layers_7_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/max_abs":0.000701904296875,"train/train/tensor_param_model_layers_77_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_40_post_attention_layernorm/std":1.0000039869839938,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_37_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn/mean":0.00164794921875,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/mean":2.1457672119140625e-06,"train/train/tensor_act_model_layers_61_mlp/max_abs":0.08740234375,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/std":9.373102443092358e-05,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/norm":9.635326349139124e-05,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/mean":3.090826794505119e-08,"train/train/tensor_act_model_layers_72_self_attn_o_proj/max_abs":0.2275390625,"train/train/layer_model_layers_59/grad/std":0.00020335726023958212,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/mean":-1.660737325437367e-09,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_input_layernorm/max_abs":4.71875,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/max_abs":0.09130859375,"train/train/tensor_act_model_layers_2_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_k_proj/mean":0.00218963623046875,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/max_abs":0.000972747802734375,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/norm":0.0009755654758879365,"train/train/tensor_param_model_layers_24_input_layernorm_weight/mean":1,"train/train/layer__model_layers_23/param/norm":17.927888322038935,"train/train/tensor_grad_model_layers_77_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_q_proj/norm":1354.6547513082342,"train/train/tensor_act_model_layers_0_mlp_down_proj/norm":90.1570072565903,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/mean":9.42964106798172e-09,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64/max_abs":1.9296875,"train/train/tensor_act_model_layers_31_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_k_proj/norm":1326.1859781387673,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/max_abs":0.00141143798828125,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_40_mlp_down_proj/mean":0.000789642333984375,"train/train/tensor_act_model_layers_76_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_v_proj/mean":-0.004741668701171875,"train/train/tensor_grad_model_layers_87_mlp_gate_proj_weight/max_abs":0.000873565673828125,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/std":0.0006506076652343894,"train/train/tensor_act_model_layers_46_self_attn/max_abs":0.232421875,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/max_abs":0.0002956390380859375,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_59_input_layernorm/max_abs":4.65625,"train/train/tensor_act_model_layers_58_self_attn_o_proj/norm":292.41655629687415,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_input_layernorm/std":1.0000029113101303,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_down_proj/mean":-0.00012159347534179688,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_86_post_attention_layernorm/mean":0.01495361328125,"train/train/tensor_act_model_layers_82_mlp_gate_proj/mean":-0.004497528076171875,"train/train/layer__model_layers_6/param/max_abs":1,"train/train/tensor_act_model_layers_27_self_attn_k_proj/norm":1303.5208480574922,"train/train/tensor_act_model_layers_49_mlp_gate_proj/std":0.22754522573982677,"train/train/tensor_act_model_layers_17_mlp/max_abs":0.09033203125,"train/train/layer_model_layers_93/grad/mean":-9.696978593300359e-08,"train/train/tensor_act_model_layers_13_post_attention_layernorm/std":1.0000027772000613,"train/train/tensor_param_model_layers_4_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/mean":0.00010728836059570312,"train/train/tensor_param_model_layers_59_mlp_gate_proj_weight/mean":2.726912498474121e-06,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_17/grad/max_abs":0.01007080078125,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/max_abs":5.5730342864990234e-06,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/max_abs":7.271766662597656e-06,"train/train/tensor_act_model_layers_35_self_attn_k_proj/mean":-0.000125885009765625,"train/train/tensor_act_model_layers_54_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/std":0.0004375890588958248,"train/train/tensor_param_model_layers_85_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp_up_proj/std":0.22534590336825835,"train/train/layer_model_layers_90/act/norm":9330.688521496611,"train/train/tensor_act_model_layers_16_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/std":5.636015742134617e-07,"train/train/tensor_act_model_layers_22_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/max_abs":0.08447265625,"train/train/tensor_param_model_layers_13_mlp_gate_proj_weight/mean":-6.866455078125e-05,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/mean":1.8766149878501892e-06,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/norm":0.055569082507429146,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/mean":1.2531876564025879e-05,"train/train/tensor_act_model_layers_66_post_attention_layernorm/max_abs":4.53125,"train/train/tensor_act_model_layers_34_post_attention_layernorm/mean":-0.0186767578125,"train/train/tensor_act_model_layers_64_mlp_up_proj/norm":1848.1199520966802,"train/train/tensor_act_model_layers_30_self_attn/mean":0.0004100799560546875,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/norm":0.23083776898948102,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/max_abs":0.0791015625,"train/train/tensor_act_model_layers_36_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_k_proj/max_abs":1.1640625,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/max_abs":0.00028228759765625,"train/train/tensor_act_model_layers_86_self_attn/norm":281.20053303639503,"train/train/tensor_act_model_layers_48_mlp_up_proj/frac_near_user_limit":0,"train/train/layer_model_layers_24/grad/mean":2.831378632026046e-07,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/max_abs":0.0849609375,"train/train/tensor_act_model_layers_51_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_44_self_attn_v_proj/std":0.23218235177744553,"train/train/tensor_act_model_layers_8_self_attn_q_proj/mean":0.00333404541015625,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_74/act/norm":9256.62478761999,"train/train/tensor_act_model_layers_39_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/mean":-1.3172626495361328e-05,"train/train/tensor_act_model_layers_44/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp/std":0.01568665504245393,"train/train/tensor_act_model_layers_76_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/max_abs":0.0010528564453125,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp/std":0.015808777883169454,"train/train/tensor_act_model_layers_92_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_k_proj/mean":-0.013580322265625,"train/train/tensor_act_model_layers_17/std":0.20020413495604472,"train/train/tensor_grad_model_layers_1_mlp_gate_proj_weight/max_abs":0.0103759765625,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_gate_proj/mean":0.0086212158203125,"train/train/tensor_act_model_layers_9_mlp_up_proj/max_abs":1.1796875,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_post_attention_layernorm/std":1.0000040812563775,"train/train/layer_model_layers_10/act/mean":-0.007230045539992196,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/mean":-1.6298145055770874e-08,"train/train/tensor_act_model_layers_50_input_layernorm/norm":5792.583007813992,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_25_mlp_gate_proj/mean":0.0021991729736328125,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/mean":5.926936864852905e-06,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/mean":4.954636096954346e-07,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_v_proj/mean":0.0018447637557983398,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/norm":2.5625,"train/train/layer_model_layers_9/grad/max_abs":0.011474609375,"train/train/tensor_act_model_layers_58_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_input_layernorm/mean":0.01055908203125,"train/train/tensor_act_model_layers_30/mean":-0.0112152099609375,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/max_abs":0.004547119140625,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/mean":-4.220008850097656e-05,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_14/act/max_abs":5.0625,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_47_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_93_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_77_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_72/act/mean":-0.002747229167393276,"train/train/tensor_act_model_layers_89_input_layernorm/norm":5792.596313477987,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_q_proj/std":0.22192982326148328,"train/train/tensor_act_model_layers_20_input_layernorm/norm":5792.53808594213,"train/train/layer__model_layers_19/param/std":0.04425515722355851,"train/train/tensor_act_model_layers_55_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/norm":0.03046051794498437,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/norm":0.10242099322209337,"train/train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp/mean":-0.0005245208740234375,"train/train/tensor_act_model_layers_26_self_attn_o_proj/norm":287.8425469270435,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/mean":-8.523929864168167e-07,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/std":5.331112628246799e-05,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/norm":0.006174645797854709,"train/train/tensor_act_model_layers_53_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_k_proj/std":0.22169083131915712,"train/train/tensor_act_model_layers_11/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_gate_proj_weight/std":0.02001953125,"train/train/layer_model_layers_88/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/std":0.0004032200398063766,"train/train/tensor_act_model_layers_51_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/max_abs":6.0498714447021484e-06,"train/train/tensor_act_model_layers_89/mean":0.005146026611328125,"train/train/tensor_act_model_layers_38_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_26/grad/mean":2.4604968517414307e-07,"train/train/tensor_act_model_layers_73_self_attn_k_proj/norm":1337.2719072070126,"train/train/tensor_act_model_layers_53_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/std":0.00012187318465057738,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/std":0.00033458578801147073,"train/train/tensor_act_model_layers_23/max_abs":1.15625,"train/train/tensor_act_model_layers_75_self_attn_k_proj/mean":-0.00870513916015625,"train/train/tensor_act_model_layers_64_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_12_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_47/grad/frac_near_user_limit":0,"train/train/layer__model_layers_54/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_74_mlp/norm":91.52036738745716,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/max_abs":6.467103958129883e-06,"train/train/tensor_act_model_layers_45_self_attn_v_proj/mean":-0.004467010498046875,"train/train/tensor_act_model_layers_37_self_attn_q_proj/max_abs":1.0234375,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_v_proj/mean":0.00567626953125,"train/train/layer__model_layers_84/param/norm":17.936390699584045,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/max_abs":0.001129150390625,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/mean":-4.652887582778931e-06,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/std":0.0011879735280358778,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/max_abs":0.0859375,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_81_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_85/param/max_abs":1,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn_v_proj/mean":-0.00753021240234375,"train/train/tensor_act_model_layers_58_post_attention_layernorm/std":1.000010516788457,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/std":0.019775390625,"train/train/layer__model_layers_12/param/mean":0.0016239065089947349,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/mean":9.678304195404053e-06,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/norm":0.26038212865043603,"train/train/tensor_act_model_layers_59_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_61/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/norm":0.03002352775423181,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_6/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_o_proj/std":0.048890123159887076,"train/train/tensor_act_model_layers_63_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56/max_abs":1.8046875,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/max_abs":0.0002231597900390625,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_o_proj/max_abs":0.23046875,"train/train/layer_model_layers_48/grad/norm":0.17463277466988017,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/mean":-5.46872615814209e-05,"train/train/tensor_act_model_layers_73_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/mean":4.682224243879318e-07,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/max_abs":6.020069122314453e-06,"train/train/layer_model_layers_66/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_post_attention_layernorm/std":1.0000038235375204,"train/train/layer_model_layers_64/act/mean":0.0018133882965360368,"train/train/tensor_act_model_layers_45_input_layernorm/max_abs":4.875,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/max_abs":0.080078125,"train/train/layer_model_layers_92/act/norm":9341.769641070998,"train/train/layer__model_layers_93/param/norm":17.93999737623657,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/max_abs":0.08447265625,"train/train/layer__model_layers_65/param/mean":0.0015817212984082108,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/max_abs":0.07275390625,"train/train/tensor_param_model_layers_4_mlp_gate_proj_weight/mean":-8.344650268554688e-05,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/mean":0.00014591217041015625,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/std":0.00033621524451061196,"train/train/tensor_act_model_layers_8_self_attn_v_proj/norm":1340.614136317916,"train/train/tensor_act_model_layers_59_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18/std":0.20679407659726154,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/max_abs":7.68899917602539e-06,"train/train/tensor_act_model_layers_39_self_attn_v_proj/norm":1315.9885424110387,"train/train/layer_model_layers_86/grad/std":0.00018084221875541327,"train/train/tensor_act_model_layers_23_input_layernorm/std":0.9990316560319558,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/mean":-1.415855876985006e-09,"train/train/tensor_act_model_layers_9_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/mean":0.00010538101196289062,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_24_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer__model_layers_48/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/norm":0.00015878820561860204,"train/train/tensor_act_model_layers_29_post_attention_layernorm/max_abs":4.78125,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/max_abs":0.00994873046875,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_92_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_49/grad/mean":-3.83683833382245e-08,"train/train/tensor_grad_model_layers_44_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp/norm":89.01714880258355,"train/train/tensor_act_model_layers_62_mlp_up_proj/norm":1830.091271538371,"train/train/layer_model_layers_12/act/norm":8958.74784599086,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/mean":1.0477378964424133e-09,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/max_abs":0.076171875,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/norm":0.00012904855718096444,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/norm":0.0011334979926506117,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/std":0.0004996135906016111,"train/train/tensor_param_model_layers_80_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/max_abs":0.00119781494140625,"train/train/tensor_act_model_layers_30_mlp_up_proj/norm":1889.0695809370006,"train/train/tensor_act_model_layers_60_mlp_up_proj/norm":1855.8081510674626,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/norm":0.10315280886148076,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/mean":-6.542541086673737e-08,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/norm":0.13075457119655107,"train/train/tensor_act_model_layers_36_mlp_gate_proj/std":0.23095958689526733,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/norm":0.1730259592061379,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/mean":1.0088086128234863e-05,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/mean":1.7713755369186401e-06,"train/train/tensor_act_model_layers_38_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp_down_proj/mean":0.00010579824447631836,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/max_abs":0.00124359130859375,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_24_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn/norm":280.0751191746162,"train/train/tensor_act_model_layers_22_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_v_proj/mean":0.0102081298828125,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/mean":0.0002918243408203125,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/mean":-0.0002994537353515625,"train/train/tensor_act_model_layers_85_self_attn_v_proj/norm":1302.54509627806,"train/train/layer__model_layers_35/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_18/param/norm":17.940466871363743,"train/train/tensor_act_model_layers_10_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/mean":0.00030517578125,"train/train/tensor_act_model_layers_36_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_32/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/mean":0.00016021728515625,"train/train/tensor_param_model_layers_90_mlp_gate_proj_weight/mean":0.0001316070556640625,"train/train/layer__model_layers_46/param/max_abs":1,"train/train/tensor_param_model_layers_56_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/mean":-8.344650268554688e-05,"train/train/tensor_param_model_layers_8_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_21/grad/max_abs":0.007598876953125,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/std":0.0003060746639262076,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/norm":0.013708100691109271,"train/train/tensor_grad_model_layers_78_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_down_proj/norm":91.51150192762687,"train/train/tensor_act_model_layers_17_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/mean":0.0003490447998046875,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_54_self_attn_o_proj/max_abs":0.255859375,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/norm":3.625,"train/train/layer_model_layers_54/grad/std":0.00020140560064767785,"train/train/tensor_act_model_layers_27_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_45/param/std":0.04423876241171064,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/std":0.00013763430090558962,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/mean":1.9744038581848145e-05,"train/train/tensor_act_model_layers_49_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_q_proj/mean":0.012481689453125,"train/train/layer__model_layers_51/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/max_abs":0.005096435546875,"train/train/tensor_act_model_layers_73_mlp_gate_proj/std":0.2270542086834783,"train/train/tensor_act_model_layers_34_mlp_up_proj/norm":1843.9006295276351,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_mlp_up_proj/max_abs":1.0859375,"train/train/tensor_act_model_layers_31_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_mlp_gate_proj_weight/std":0.020263671875,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_60/param/std":0.04423174475338259,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/norm":0.028756692978038567,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/max_abs":0.005584716796875,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/norm":0.00020255927985347288,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/mean":-3.103195922449231e-09,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/mean":0.0002593994140625,"train/train/tensor_act_model_layers_0_self_attn/mean":-0.000823974609375,"train/train/tensor_act_model_layers_87_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_input_layernorm/mean":-0.058837890625,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/norm":2.578125,"train/train/layer__model_layers_28/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/max_abs":0.080078125,"train/train/tensor_act_model_layers_93_post_attention_layernorm/std":1.0000016260878932,"train/train/layer_model_layers_59/grad/mean":-2.359488398539899e-07,"train/train/layer_model_layers_8/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/max_abs":0.08154296875,"train/train/tensor_act_model_layers_91_self_attn_q_proj/mean":0.0089111328125,"train/train/tensor_act_model_layers_43_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/mean":1.1682510375976562e-05,"train/train/layer_model_layers_40/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_57_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_q_proj/max_abs":1.2265625,"train/train/layer_model_layers_25/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_input_layernorm/max_abs":5.09375,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/max_abs":0.08203125,"train/train/tensor_act_model_layers_71_mlp_down_proj/mean":0.000637054443359375,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/std":7.01137800193993e-05,"train/train/layer_model_layers_55/act/mean":8.631018655640739e-05,"train/train/tensor_act_model_layers_53_self_attn/std":0.04638846666432787,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/std":0.0001555172073604305,"train/train/layer_model_layers_28/act/std":0.41723449205250634,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_46_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_o_proj/std":0.05072163776812896,"train/train/tensor_grad_model_layers_63_mlp_gate_proj_weight/std":8.466944839563945e-05,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/std":0.00011800537673323706,"train/train/tensor_param_model_layers_38_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/std":4.126962488851618e-07,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/max_abs":3.993511199951172e-06,"train/train/tensor_act_model_layers_18_self_attn_v_proj/norm":1250.3994033147164,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/std":5.951716104920962e-05,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp_down_proj/mean":0.00026679039001464844,"train/train/tensor_act_model_layers_58_self_attn/max_abs":0.2275390625,"train/train/tensor_act_model_layers_16_self_attn_q_proj/std":0.21777489794254265,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/std":8.454913350222925e-07,"train/train/layer__model_layers_26/param/std":0.04424032398148525,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/mean":5.1746610552072525e-08,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/mean":-6.757676601409912e-06,"train/train/layer__model_layers_57/param/mean":0.0015871878161259263,"train/train/tensor_act_model_layers_52_self_attn_k_proj/std":0.22217750170867717,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/mean":-1.1119991540908813e-06,"train/train/tensor_act_model_layers_15_mlp_down_proj/std":0.016205620022610872,"train/train/tensor_act_model_layers_10_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_38/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_post_attention_layernorm/norm":5792.587280274388,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/mean":0.0002956390380859375,"train/train/tensor_grad_model_layers_90_mlp_gate_proj_weight/std":6.802436227327297e-05,"train/train/tensor_act_model_layers_38_mlp_up_proj/max_abs":1.234375,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/std":0.0003083039845381073,"train/train/tensor_act_model_layers_10_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/mean":1.4031538739800453e-06,"train/train/tensor_act_model_layers_11_self_attn_v_proj/std":0.22754527357735396,"train/train/layer__model_layers_55/param/norm":17.931013363286944,"train/train/tensor_param_model_layers_37_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp_gate_proj/mean":0.00868988037109375,"train/train/tensor_act_model_layers_24/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/max_abs":0.07958984375,"train/train/layer_model_layers_73/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_33/grad/norm":0.23552444627100586,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/max_abs":0.0869140625,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_18_self_attn_v_proj/max_abs":1.1015625,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/norm":0.023823578197488857,"train/train/layer_model_layers_70/grad/norm":0.16277166036972052,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_post_attention_layernorm/std":1.0000015453830993,"train/train/tensor_param_model_layers_92_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/norm":2.53125,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_4_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/max_abs":0.015625,"train/train/tensor_act_model_layers_21_mlp_gate_proj/mean":0.0010347366333007812,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_78_mlp_down_proj/std":0.015717479848034024,"train/train/tensor_act_model_layers_63_self_attn/norm":282.9645421103391,"train/train/layer_model_layers_82/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/max_abs":0.0002593994140625,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_q_proj/mean":0.012420654296875,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/max_abs":0.005340576171875,"train/train/tensor_param_model_layers_67_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_k_proj/std":0.22461327000018874,"train/train/tensor_act_model_layers_1_self_attn_v_proj/max_abs":1.1328125,"train/train/tensor_act_model_layers_93_mlp_up_proj/mean":0.002994537353515625,"train/train/tensor_act_model_layers_88_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/std":0.00024147133817111185,"train/train/tensor_act_model_layers_35/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_v_proj/norm":1336.5931284898245,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/max_abs":0.0011444091796875,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/std":0.00010135645411366083,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/mean":1.928128767758608e-09,"train/train/tensor_act_model_layers_60_self_attn_v_proj/mean":0.00395965576171875,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/std":0.019775390625,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/mean":-0.00010013580322265625,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/mean":-0.0001811981201171875,"train/train/tensor_param_model_layers_78_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/norm":0.0001338734652169925,"train/train/tensor_act_model_layers_15_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_38/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_89_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_26_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/mean":0.0002918243408203125,"train/train/tensor_grad_model_layers_60_mlp_gate_proj_weight/norm":0.028853494599614275,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_q_proj/std":0.22656263976257224,"train/train/tensor_act_model_layers_70/norm":2425.54679512907,"train/train/tensor_act_model_layers_12_self_attn/std":0.04834108530934276,"train/train/tensor_grad_model_layers_28_mlp_gate_proj_weight/max_abs":0.00150299072265625,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/norm":0.03291668473416247,"train/train/layer_model_layers_13/grad/norm":0.38866600745852015,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/mean":-5.936622619628906e-05,"train/train/tensor_act_model_layers_5_mlp_down_proj/mean":-0.00051116943359375,"train/train/tensor_param_model_layers_75_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/std":8.772271800294685e-07,"train/train/tensor_grad_model_layers_13_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/mean":-0.0001678466796875,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/mean":-7.581710815429688e-05,"train/train/tensor_act_model_layers_36_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn/norm":289.07544213341504,"train/train/tensor_act_model_layers_9_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/max_abs":0.00909423828125,"train/train/tensor_act_model_layers_20_input_layernorm/mean":-0.0567626953125,"train/train/tensor_act_model_layers_89_input_layernorm/mean":0.0095062255859375,"train/train/tensor_act_model_layers_43_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_k_proj/norm":1341.940482212343,"train/train/tensor_act_model_layers_88_input_layernorm/norm":5792.594726567984,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/mean":-0.0007786750793457031,"train/train/tensor_param_model_layers_72_mlp_gate_proj_weight/max_abs":0.0859375,"train/train/tensor_act_model_layers_87_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_input_layernorm/mean":-0.006252288818359375,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/std":5.901535961280191e-07,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/norm":2.5625,"train/train/layer__model_layers_58/param/max_abs":1,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_79/grad/std":0.00019611560600253736,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/norm":2.5625,"train/train/layer_model_layers_42/act/mean":-0.0018882410866873606,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/max_abs":0.0089111328125,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/max_abs":0.08154296875,"train/train/tensor_param_model_layers_30_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_84_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/mean":2.8349459171295166e-06,"train/train/tensor_param_model_layers_54_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_33_mlp_gate_proj_weight/max_abs":0.0849609375,"train/train/tensor_act_model_layers_33_self_attn/norm":273.46574315856384,"train/train/tensor_act_model_layers_58_post_attention_layernorm/norm":5792.588012698517,"train/train/tensor_act_model_layers_11_self_attn_o_proj/norm":287.34211252400235,"train/train/tensor_act_model_layers_19/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_mlp_gate_proj_weight/max_abs":0.00433349609375,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp_up_proj/std":0.22046668373489894,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_11_mlp/norm":91.45940642016465,"train/train/tensor_act_model_layers_61_mlp/mean":-6.3478946685791016e-06,"train/train/tensor_grad_model_layers_17_mlp_gate_proj_weight/max_abs":0.00179290771484375,"train/train/tensor_param_model_layers_53_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_24_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_27_self_attn_q_proj/norm":1318.849896765957,"train/train/tensor_act_model_layers_31_mlp_down_proj/mean":0.0007619857788085938,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/max_abs":0.00040435791015625,"train/train/tensor_act_model_layers_35_mlp_gate_proj/norm":1857.7308371337763,"train/train/tensor_act_model_layers_29_self_attn/norm":277.2471470731721,"train/train/layer_model_layers_70/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/mean":1.0192161425948143e-07,"train/train/tensor_grad_model_layers_28_mlp_gate_proj_weight/std":0.00012730692237574752,"train/train/tensor_act_model_layers_16_self_attn_q_proj/max_abs":1.2109375,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/norm":0.08985867583643435,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_84_mlp/std":0.015518404410028785,"train/train/layer__model_layers_82/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_7_mlp_gate_proj_weight/max_abs":0.08154296875,"train/train/tensor_act_model_layers_92_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/mean":1.0440126061439514e-06,"train/train/tensor_act_model_layers_40_mlp_up_proj/norm":1830.2381982503275,"train/train/tensor_grad_model_layers_5_mlp_gate_proj_weight/norm":0.11518910698848625,"train/train/layer__model_layers_64/param/norm":17.939541477905728,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/norm":0.09004619843180646,"train/train/tensor_param_model_layers_81_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_input_layernorm/mean":-0.0084381103515625,"train/train/tensor_act_model_layers_49_self_attn_v_proj/norm":1335.842119274276,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/mean":2.0035076886415482e-07,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/mean":-8.225440979003906e-06,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_o_proj/std":0.05090791166419868,"train/train/tensor_act_model_layers_50_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/max_abs":0.08447265625,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/max_abs":0.08740234375,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/std":0.00033021680339121964,"train/train/tensor_act_model_layers_2_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_0/param/norm":17.93189835225554,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/mean":0.000133514404296875,"train/train/tensor_act_model_layers_64_self_attn_k_proj/norm":1312.9795741881162,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_6_self_attn_o_proj/std":0.04437365704374491,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/mean":-1.2233853340148926e-05,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/std":0.00012554320096285138,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/std":9.0264555046618e-05,"train/train/tensor_grad_model_layers_49_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn/mean":-0.0006258487701416016,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/norm":0.04685147012787657,"train/train/tensor_act_model_layers_50_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_92/grad/std":0.0001686054776396271,"train/train/tensor_act_model_layers_17_mlp/mean":0.0005807876586914062,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_39/act/max_abs":4.84375,"train/train/tensor_act_model_layers_61_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_78/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_12_mlp_gate_proj_weight/max_abs":0.0025177001953125,"train/train/tensor_act_model_layers_53/std":0.35938659845451393,"train/train/tensor_grad_model_layers_57_mlp_gate_proj_weight/max_abs":0.000820159912109375,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/mean":5.559995770454407e-07,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_89/param/mean":0.001592520060665708,"train/train/tensor_grad_model_layers_64_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_v_proj/norm":1337.0290333359198,"train/train/layer__model_layers_33/param/mean":0.0015494187424967701,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/norm":0.15554957756787052,"train/train/layer_model_layers_48/grad/std":0.00021554072859379994,"train/train/layer_model_layers_28/grad/mean":1.112470082218022e-06,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/max_abs":0.07958984375,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/mean":3.337860107421875e-05,"train/train/layer__model_layers_91/param/std":0.04425135112856615,"train/train/layer_model_layers_80/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_v_proj/std":0.2160679849238995,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/mean":-1.4454126358032227e-05,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/max_abs":0.083984375,"train/train/tensor_act_model_layers_76_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn_v_proj/std":0.22925035410630853,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/max_abs":9.72747802734375e-05,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_85/param/std":0.044249431595340796,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/std":3.1195557854192706e-05,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/max_abs":0.07958984375,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/max_abs":0.0002765655517578125,"train/train/tensor_act_model_layers_3_mlp/norm":88.42887849898709,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/max_abs":3.844499588012695e-06,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/max_abs":0.0888671875,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/mean":0.00014781951904296875,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/max_abs":5.8710575103759766e-06,"train/train/layer_model_layers_16/grad/max_abs":0.009521484375,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_86_mlp_down_proj/max_abs":0.0830078125,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/mean":4.5262277126312256e-07,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_29_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_83/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_self_attn_q_proj/max_abs":1.328125,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/std":0.0004381883341643351,"train/train/tensor_act_model_layers_86_self_attn_v_proj/std":0.22022435755594963,"train/train/tensor_act_model_layers_53_self_attn_v_proj/max_abs":1.015625,"train/train/tensor_param_model_layers_37_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/norm":0.027126676771310965,"train/train/tensor_act_model_layers_38_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_37_post_attention_layernorm/norm":5792.574584962866,"train/train/tensor_param_model_layers_71_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/max_abs":0.00146484375,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/std":7.881279765028646e-05,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/norm":0.21282958109193506,"train/train/tensor_act_model_layers_29_input_layernorm/mean":-0.03485107421875,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_mlp_gate_proj_weight/mean":-2.5634653866291046e-07,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/mean":4.1583552956581116e-07,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_k_proj/max_abs":1.09375,"train/train/tensor_act_model_layers_2_self_attn/mean":0.0016460418701171875,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/std":0.00020903864007851852,"train/train/layer__model_layers_39/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/mean":-0.000152587890625,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/norm":2.53125,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_mlp_gate_proj/std":0.22656613829482564,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/mean":-9.695068001747131e-07,"train/train/tensor_act_model_layers_46_self_attn_o_proj/max_abs":0.232421875,"train/train/layer_model_layers_86/grad/mean":1.5396643691520423e-07,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/max_abs":0.09521484375,"train/train/tensor_act_model_layers_65_mlp_up_proj/mean":0.001461029052734375,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/max_abs":0.0013885498046875,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/std":7.393689311170637e-05,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/max_abs":0.0732421875,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/mean":-1.3443641364574432e-06,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/max_abs":0.00013446807861328125,"train/train/tensor_act_model_layers_58_self_attn/norm":292.41655629687415,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_22_mlp_up_proj/mean":0.0113525390625,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/mean":2.5193003239110112e-09,"train/train/tensor_act_model_layers_69_self_attn_q_proj/std":0.22022243101177144,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/norm":0.00022841346024228331,"train/train/tensor_act_model_layers_11_post_attention_layernorm/mean":-0.06378173828125,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/mean":1.0772055247798562e-08,"train/train/tensor_act_model_layers_92_mlp/std":0.015274935047781182,"train/train/tensor_act_model_layers_76_self_attn_o_proj/max_abs":0.251953125,"train/train/tensor_param_model_layers_35_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_78_self_attn_k_proj/mean":0.0034847259521484375,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/norm":0.00024418439323773805,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_down_proj/max_abs":0.0849609375,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/norm":0.3041204408679412,"train/train/tensor_act_model_layers_90_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_64/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/norm":3.609375,"train/train/layer_model_layers_38/grad/norm":0.20992295474495376,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/norm":0.0007060694468737619,"train/train/tensor_act_model_layers_29_self_attn_q_proj/mean":-0.0027179718017578125,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_62_mlp_down_proj/norm":89.86351787799238,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/max_abs":0.0908203125,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/std":0.000247488125372486,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/max_abs":0.003631591796875,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/max_abs":0.0128173828125,"train/train/tensor_act_model_layers_84_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/mean":-4.384666681289673e-06,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/norm":0.11632602666700627,"train/train/tensor_param_model_layers_53_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_7_mlp_gate_proj_weight/mean":-0.000308990478515625,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/mean":1.0459189070388675e-10,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/norm":0.00012125665163592247,"train/train/tensor_act_model_layers_66_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_post_attention_layernorm/max_abs":4.46875,"train/train/tensor_param_model_layers_38_mlp_gate_proj_weight/std":0.020263671875,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_50_mlp_gate_proj_weight/mean":1.8568243831396103e-07,"train/train/tensor_act_model_layers_6_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/max_abs":0.000141143798828125,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/mean":-8.381903171539307e-09,"train/train/tensor_act_model_layers_32_self_attn_q_proj/mean":-0.0030307769775390625,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_post_attention_layernorm/norm":5792.579467776086,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_8_mlp_gate_proj/std":0.2248547406263061,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_o_proj/std":0.04913560850043175,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/mean":0.00020503997802734375,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/max_abs":4.500150680541992e-06,"train/train/tensor_act_model_layers_13_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/mean":0.00021076202392578125,"train/train/tensor_act_model_layers_72_mlp/mean":0.00023245811462402344,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/mean":7.560010999441147e-07,"train/train/tensor_act_model_layers_68_mlp_up_proj/mean":-0.00690460205078125,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/norm":0.0029880258507138163,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/std":5.176267180634271e-07,"train/train/tensor_act_model_layers_45_mlp_up_proj/max_abs":1.03125,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/norm":0.007393918589881181,"train/train/tensor_act_model_layers_51_self_attn_k_proj/max_abs":1.03125,"train/train/tensor_act_model_layers_29_post_attention_layernorm/std":1.00000128708697,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/norm":0.1432604206549063,"train/train/tensor_act_model_layers_41_self_attn_o_proj/norm":289.07544213341504,"train/train/tensor_grad_model_layers_48_mlp_gate_proj_weight/std":8.765164098863506e-05,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/mean":-3.862660378217697e-07,"train/train/tensor_act_model_layers_42_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/mean":1.862645149230957e-07,"train/train/tensor_act_model_layers_45_self_attn_q_proj/norm":1307.424113998556,"train/train/tensor_param_model_layers_83_mlp_gate_proj_weight/mean":9.965896606445312e-05,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn/mean":0.00043451786041259766,"train/train/tensor_act_model_layers_84_self_attn_o_proj/mean":0.00147247314453125,"train/train/tensor_act_model_layers_58_mlp_gate_proj/std":0.22216956563088475,"train/train/tensor_param_model_layers_85_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/max_abs":0.08203125,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn_q_proj/norm":1272.2222489686683,"train/train/tensor_act_model_layers_37_mlp/max_abs":0.0986328125,"train/train/tensor_act_model_layers_93_mlp_up_proj/max_abs":1.125,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/mean":2.47955322265625e-05,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_q_proj/norm":1287.7366719305935,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp_up_proj/max_abs":1.2265625,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/norm":0.0019415909721965212,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_81_self_attn/norm":293.8579930583087,"train/train/layer__model_layers_62/param/max_abs":1,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/std":0.02001953125,"train/train/layer_model_layers_6/grad/mean":-5.363390996017713e-07,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/mean":5.711626727133989e-10,"train/train/layer__model_layers_57/param/max_abs":1,"train/train/tensor_act_model_layers_24_mlp_up_proj/norm":1859.361654072554,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/max_abs":8.64267349243164e-06,"train/train/tensor_param_model_layers_92_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/max_abs":0.080078125,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/norm":0.00010564983395323431,"train/train/tensor_act_model_layers_63_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_o_proj/norm":295.60741299078444,"train/train/layer_model_layers_28/act/max_abs":4.65625,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/max_abs":0.0067138671875,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_53/act/norm":9145.913811745791,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/max_abs":0.0771484375,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/norm":0.002402895945377452,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/mean":-7.0035457611083984e-06,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/mean":-9.000301361083984e-06,"train/train/tensor_act_model_layers_78_mlp/std":0.015717479848034024,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_46/grad/max_abs":0.004119873046875,"train/train/tensor_act_model_layers_74_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_49/act/frac_near_user_limit":0,"train/train/layer_model_layers_43/act/mean":-0.0012163434709821428,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/std":0.02001953125,"train/train/layer_model_layers_12/grad/mean":1.6403707055238405e-06,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_86_mlp_gate_proj_weight/mean":1.1463998816907406e-07,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/std":4.562365034280197e-07,"train/train/tensor_grad_model_layers_27_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_v_proj/norm":1315.961867426427,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/mean":-1.6728881746530533e-07,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/std":5.013509524081533e-05,"train/train/tensor_act_model_layers_79_mlp_down_proj/std":0.015488971304149621,"train/train/tensor_param_model_layers_88_mlp_gate_proj_weight/max_abs":0.09130859375,"train/train/tensor_act_model_layers_54_self_attn/max_abs":0.255859375,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/max_abs":0.006011962890625,"train/train/tensor_act_model_layers_62_mlp_gate_proj/max_abs":1.15625,"train/train/tensor_param_model_layers_65_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_gate_proj_weight/norm":0.11774515346911867,"train/train/tensor_act_model_layers_39_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/mean":0.00022792816162109375,"train/train/tensor_grad_model_layers_73_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/norm":0.024915953264474634,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/max_abs":0.0038909912109375,"train/train/tensor_param_model_layers_34_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/std":0.0005107601882956524,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/max_abs":0.07666015625,"train/train/tensor_act_model_layers_29_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/mean":-4.982948303222656e-05,"train/train/tensor_act_model_layers_7_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_v_proj/norm":1298.188363830075,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_27/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_36_self_attn/frac_near_dtype_limit":0,"train/train/layer__model_layers_32/param/max_abs":1,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_19_self_attn_o_proj/norm":287.18764228872,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/mean":0.0001926422119140625,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_k_proj/mean":0.0007123947143554688,"train/train/tensor_act_model_layers_49_input_layernorm/mean":-0.000152587890625,"train/train/tensor_act_model_layers_91_self_attn_q_proj/std":0.22095274944771212,"train/train/tensor_act_model_layers_89_self_attn_o_proj/std":0.04742491269526409,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/max_abs":2.0384788513183594e-05,"train/train/tensor_act_model_layers_29_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_input_layernorm/max_abs":5,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/std":0.0005305792802410225,"train/train/tensor_act_model_layers_73_self_attn_v_proj/std":0.2299907286191901,"train/train/tensor_act_model_layers_11_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_post_attention_layernorm/max_abs":4.65625,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_28_mlp_gate_proj_weight/max_abs":0.0849609375,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/std":9.636912725954772e-05,"train/train/tensor_act_model_layers_7_mlp/max_abs":0.091796875,"train/train/tensor_act_model_layers_75_post_attention_layernorm/mean":0.00380706787109375,"train/train/tensor_act_model_layers_69_self_attn/mean":-0.0021514892578125,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/mean":3.6716461181640625e-05,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_50/grad/std":0.00021419657982072062,"train/train/tensor_act_model_layers_29_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_q_proj/max_abs":1.03125,"train/train/layer_model_layers_56/grad/mean":1.5734030186618546e-07,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/mean":-0.00011587142944335938,"train/train/tensor_param_model_layers_59_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_post_attention_layernorm/max_abs":5,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/max_abs":6.258487701416016e-06,"train/train/layer__model_layers_23/param/mean":0.001582111471714728,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/max_abs":0.00421142578125,"train/train/layer_model_layers_60/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_post_attention_layernorm/norm":5792.5946044941265,"train/train/tensor_act_model_layers_61_self_attn_q_proj/norm":1288.5868955329966,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_input_layernorm/std":1.0000038485667186,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/max_abs":0.07958984375,"train/train/tensor_act_model_layers_9_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/norm":0.0020760506381920685,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/max_abs":3.904104232788086e-06,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp_down_proj/max_abs":0.07763671875,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_62_post_attention_layernorm/std":1.000007954762674,"train/train/tensor_act_model_layers_29_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/std":1.1579654987435492e-06,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/mean":1.7955899238586426e-06,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/mean":-0.0002536773681640625,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/mean":1.648440957069397e-07,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/norm":0.0027005146969026885,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn_k_proj/norm":1358.1570739988274,"train/train/tensor_act_model_layers_2_self_attn_o_proj/norm":180.00567329164943,"train/train/tensor_act_model_layers_66_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/std":5.163990544264699e-07,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/norm":0.024486890886334177,"train/train/tensor_grad_model_layers_12_mlp_gate_proj_weight/norm":0.07441965317437976,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_embed_tokens/norm":117.15984568360321,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_55_mlp_gate_proj_weight/mean":1.0855728760361671e-07,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_28_post_attention_layernorm/mean":-0.033233642578125,"train/train/tensor_param_model_layers_91_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn/norm":303.2164695260724,"train/train/tensor_act_model_layers_35_mlp_down_proj/std":0.015808779458642157,"train/train/tensor_act_model_layers_62_mlp_gate_proj/norm":1836.041603091663,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/max_abs":0.078125,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_lm_head/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_mlp_gate_proj_weight/mean":3.585591912269592e-08,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_embed_tokens_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/norm":0.002281367572439323,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/max_abs":0.07958984375,"train/train/tensor_act_model_layers_11_mlp_down_proj/mean":0.00015461444854736328,"train/train/tensor_act_model_layers_72_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_38/param/max_abs":1,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/std":0.0009217727040756839,"train/train/tensor_act_model_layers_83/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_v_proj/max_abs":1.1484375,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/std":0.0003730443636177762,"train/train/tensor_act_model_layers_83_mlp_up_proj/mean":-0.0023326873779296875,"train/train/layer_model_layers_46/grad/mean":2.2012375690356804e-07,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_34_mlp_gate_proj_weight/norm":0.03960462131603826,"train/train/tensor_act_model_layers_59_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn_k_proj/max_abs":1.046875,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/norm":0.024845732123999223,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_31_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_9_self_attn_o_proj/norm":270.1541706032064,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/norm":0.026100434115332926,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/max_abs":5.364418029785156e-06,"train/train/tensor_grad_model_layers_85_mlp_gate_proj_weight/std":7.43458088196995e-05,"train/train/tensor_act_model_layers_15_self_attn_o_proj/mean":-0.001697540283203125,"train/train/tensor_param_model_layers_90_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_input_layernorm/max_abs":4.65625,"train/train/tensor_act_model_layers_57_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_81/act/norm":9282.196296429805,"train/train/tensor_param_model_layers_50_mlp_gate_proj_weight/max_abs":0.091796875,"train/train/layer_model_layers_69/grad/norm":0.16599033148329062,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp/mean":-7.270276546478271e-05,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/norm":0.09286911698573144,"train/train/layer_model_layers_57/grad/max_abs":0.0036163330078125,"train/train/layer__model_layers_27/param/mean":0.0015397808295144306,"train/train/tensor_act_model_layers_74/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_72_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/max_abs":0.1044921875,"train/train/tensor_act_model_layers_85_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_mlp_up_proj/std":0.22900612406581924,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/mean":-5.862675607204437e-07,"train/train/tensor_act_model_layers_6_input_layernorm/norm":5792.272216801212,"train/train/layer__model_layers_62/param/mean":0.001594555359362812,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_17_mlp_up_proj/std":0.23071516265456835,"train/train/tensor_param_model_layers_57_mlp_gate_proj_weight/norm":3.640625,"train/train/layer_model_layers_44/act/norm":9112.904298413981,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_46/param/mean":0.0015056925519208268,"train/train/tensor_act_model_layers_77_mlp_down_proj/max_abs":0.08203125,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp_gate_proj/mean":-0.001434326171875,"train/train/tensor_act_model_layers_38_self_attn_o_proj/max_abs":0.24609375,"train/train/tensor_act_model_layers_3_self_attn_q_proj/std":0.22802931375796154,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/max_abs":0.000347137451171875,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp/max_abs":0.08935546875,"train/train/tensor_act_model_layers_78_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/std":0.0001352627143110529,"train/train/tensor_param_model_layers_20_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_mlp_gate_proj/std":0.22998253869507532,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/mean":-1.5079975128173828e-05,"train/train/tensor_param_model_layers_36_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_32_post_attention_layernorm/mean":-0.02587890625,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/mean":-2.4668406695127487e-07,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/std":0.00010743866305366419,"train/train/tensor_act_model_layers_72/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/std":0.020263671875,"train/train/tensor_param_model_layers_69_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_mlp_gate_proj_weight/mean":3.543682396411896e-07,"train/train/tensor_act_model_layers_66_mlp_up_proj/norm":1845.5445182928904,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/max_abs":0.0771484375,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_32_self_attn_q_proj/max_abs":1.1640625,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/max_abs":0.07763671875,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/mean":-1.3232231140136719e-05,"train/train/tensor_act_model_layers_72_post_attention_layernorm/max_abs":4.59375,"train/train/tensor_act_model_layers_0_self_attn_k_proj/std":0.22778359832285708,"train/train/tensor_param_model_layers_58_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_q_proj/max_abs":1.046875,"train/train/tensor_act_model_layers_29_mlp_down_proj/mean":-0.00024116039276123047,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/mean":0.00017261505126953125,"train/train/layer_model_layers_57/grad/norm":0.1565534214438069,"train/train/tensor_param_model_layers_36_mlp_gate_proj_weight/mean":-0.00010728836059570312,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/mean":-1.6461126506328583e-07,"train/train/layer__model_layers_31/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp_up_proj/max_abs":1.1484375,"train/train/tensor_param_model_layers_18_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp/std":0.0151829737139752,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/max_abs":0.09130859375,"train/train/layer__model_layers_59/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_q_proj/norm":1284.1350518845327,"train/train/tensor_act_model_layers_21_self_attn_o_proj/max_abs":0.224609375,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/max_abs":0.08935546875,"train/train/tensor_act_model_layers_69_mlp_down_proj/frac_near_dtype_limit":0,"train/train/global/act/std":0.4187010004306271,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/mean":7.545604603365064e-07,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_post_attention_layernorm/norm":5792.590454102873,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/std":0.00038203927346681205,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/std":3.969645704655046e-07,"train/train/tensor_act_model_layers_12_self_attn/max_abs":0.2392578125,"train/train/layer__model_layers_20/param/mean":0.0015149287612129486,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_26_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer__model_layers_44/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_post_attention_layernorm/std":1.0000030733597733,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/std":6.881732760730428e-05,"train/train/tensor_act_model_layers_5_mlp/mean":-0.00051116943359375,"train/train/tensor_act_model_layers_36_self_attn_k_proj/norm":1282.6293008845394,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_77/act/norm":9268.575047867527,"train/train/tensor_act_model_layers_47_mlp/norm":91.83005063890906,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/max_abs":0.0002460479736328125,"train/train/tensor_act_model_layers_38_input_layernorm/mean":-0.007183074951171875,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/mean":0.00028228759765625,"train/train/tensor_act_model_layers_25_post_attention_layernorm/norm":5792.561523448685,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_mlp_gate_proj_weight/std":0.00010922678193167074,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/mean":-0.0003566741943359375,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/std":0.0014212515594919158,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/norm":0.03430870773121028,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/norm":3.65625,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/max_abs":0.07470703125,"train/train/tensor_grad_model_layers_9_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_v_proj/max_abs":1.1796875,"train/train/tensor_grad_model_layers_33_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn/max_abs":0.20703125,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/mean":0.0003948211669921875,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_k_proj/max_abs":1.0546875,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/mean":1.7900019884109497e-06,"train/train/layer__model_layers_36/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/norm":0.20558786166981455,"train/train/tensor_param_model_layers_53_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/max_abs":0.000701904296875,"train/train/tensor_param_model_layers_0_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_7_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_1/grad/mean":2.767887785914536e-06,"train/train/layer_model_layers_73/act/std":0.42700977477280344,"train/train/tensor_act_model_layers_60_self_attn_o_proj/std":0.04925852194229224,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/mean":-2.6472844183444977e-07,"train/train/tensor_act_model_layers_47_self_attn_k_proj/std":0.22754489907568068,"train/train/tensor_act_model_layers_42_self_attn_v_proj/norm":1306.0647580526795,"train/train/tensor_act_model_layers_11_self_attn_q_proj/max_abs":1.0625,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/norm":0.0949399452120597,"train/train/tensor_act_model_layers_67_self_attn_v_proj/max_abs":1.1640625,"train/train/tensor_grad_model_layers_53_mlp_gate_proj_weight/norm":0.030248631805027577,"train/train/tensor_act_model_layers_8_self_attn/max_abs":0.25,"train/train/tensor_act_model_layers_71_self_attn_k_proj/norm":1328.4608101714555,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/max_abs":0.09033203125,"train/train/layer_model_layers_22/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_28/grad/std":0.00030133773579682483,"train/train/tensor_param_model_layers_21_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_post_attention_layernorm/norm":5792.57836914133,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/norm":0.11487588238299405,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/max_abs":0.00494384765625,"train/train/tensor_act_model_layers_36_mlp_up_proj/norm":1907.0195630797534,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/mean":0.000244140625,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/max_abs":0.0004024505615234375,"train/train/tensor_param_model_layers_92_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_mlp_gate_proj_weight/norm":0.029364190649190294,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_mlp_gate_proj_weight/norm":0.024333133076688565,"train/train/tensor_param_model_layers_12_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_down_proj/mean":-0.0003509521484375,"train/train/layer_model_layers_67/act/mean":0.0008636116981506348,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/mean":-2.644956111907959e-06,"train/train/layer__model_layers_65/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_mlp_up_proj/max_abs":1.2265625,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/norm":2.59375,"train/train/tensor_act_model_layers_27_self_attn_v_proj/mean":0.00701141357421875,"train/train/tensor_act_model_layers_17_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/mean":7.175549399107695e-08,"train/train/tensor_act_model_layers_2_self_attn_k_proj/norm":1318.8604908996635,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/max_abs":9.655952453613281e-06,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/norm":0.1470837347585299,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/max_abs":4.410743713378906e-06,"train/train/tensor_param_model_layers_15_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_42/param/norm":17.92524625005191,"train/train/tensor_act_model_layers_61/std":0.3872155323639755,"train/train/tensor_act_model_layers_48_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/mean":2.1223968360573053e-08,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/norm":0.0016738220451760902,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_input_layernorm/max_abs":4.84375,"train/train/layer__model_layers_13/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_post_attention_layernorm/max_abs":4.5,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp_up_proj/max_abs":1.125,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/mean":1.482476363889873e-09,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/norm":0.00010162282544681524,"train/train/layer_model_layers_9/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_mlp_down_proj/std":0.015137602987144554,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/norm":2.546875,"train/train/layer__model_layers_29/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58/norm":2188.8255882995063,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/max_abs":0.08642578125,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_54_input_layernorm/norm":5792.587036136705,"train/train/tensor_act_model_layers_49_self_attn/std":0.04968509503870688,"train/train/tensor_act_model_layers_23_self_attn_o_proj/std":0.04883140046339214,"train/train/tensor_act_model_layers_92_mlp_gate_proj/std":0.22534597119564892,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/mean":-3.257300704717636e-07,"train/train/tensor_act_model_layers_36_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_31_mlp_gate_proj/std":0.22436801755090058,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_22_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_self_attn_k_proj/norm":1259.9479528404806,"train/train/tensor_act_model_layers_19_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_gate_proj/std":0.22436815245414976,"train/train/layer__model_layers_31/param/max_abs":1,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/mean":7.05718994140625e-05,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/mean":-2.2497260943055153e-07,"train/train/tensor_param_model_layers_78_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/mean":5.125999450683594e-05,"train/train/tensor_param_model_layers_72_mlp_gate_proj_weight/mean":-4.220008850097656e-05,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/mean":0.0003223419189453125,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_44/act/std":0.42051607193293217,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/max_abs":2.5272369384765625e-05,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_81_self_attn_k_proj/mean":0.0016832351684570312,"train/train/tensor_grad_model_layers_49_mlp_gate_proj_weight/mean":6.551854312419891e-07,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/norm":3.65625,"train/train/tensor_act_model_layers_14_mlp_down_proj/norm":91.64584453358795,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/mean":5.122274160385132e-08,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/max_abs":0.0001220703125,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/norm":0.02991831048595211,"train/train/tensor_act_model_layers_46_mlp_up_proj/mean":-0.00563812255859375,"train/train/layer_model_layers_13/act/max_abs":4.84375,"train/train/tensor_param_model_layers_55_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_78/act/mean":-0.0007349167551313128,"train/train/layer__model_layers_38/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_input_layernorm/norm":5792.392089850275,"train/train/tensor_act_model_layers_2_mlp_up_proj/mean":-0.00423431396484375,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/norm":0.0020347897012801715,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/mean":5.4836273193359375e-05,"train/train/layer_model_layers_7/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_input_layernorm/norm":5792.587036134421,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/norm":0.09717931633252863,"train/train/layer_model_layers_5/grad/std":0.0008101574025083183,"train/train/tensor_act_model_layers_57_mlp_down_proj/mean":-6.763637065887451e-05,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_74/act/std":0.4270909073118596,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_43/act/max_abs":4.875,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/std":3.649026566992955e-07,"train/train/tensor_act_model_layers_8_mlp_down_proj/norm":94.78675581007263,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/mean":3.748573362827301e-07,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/max_abs":0.08251953125,"train/train/layer_model_layers_45/act/mean":-0.003563499676861933,"train/train/layer__model_layers_93/param/mean":0.0016780740199334536,"train/train/tensor_param_model_layers_11_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_67/std":0.4116303808698977,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_up_proj/max_abs":1.03125,"train/train/tensor_param_model_layers_45_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/norm":2.53125,"train/train/tensor_act_model_layers_15_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/max_abs":0.078125,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54/std":0.36280518450324983,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/norm":0.002738846375680728,"train/train/tensor_act_model_layers_2_post_attention_layernorm/max_abs":5.0625,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/max_abs":0.00125885009765625,"train/train/tensor_act_model_layers_60_mlp_down_proj/std":0.015244750942809374,"train/train/tensor_act_model_layers_80_mlp_gate_proj/mean":0.0005960464477539062,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/norm":0.03378967108205999,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/std":8.035953420712252e-05,"train/train/layer__model_layers_66/param/max_abs":1,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/mean":1.5422701835632324e-06,"train/train/tensor_act_model_layers_17_mlp_gate_proj/max_abs":1.171875,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/mean":3.932509571313858e-07,"train/train/tensor_act_model_layers_89_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/norm":0.04436747903622273,"train/train/tensor_act_model_layers_73_mlp_up_proj/norm":1832.4525882544674,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/mean":0.00010061264038085938,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/max_abs":0.00421142578125,"train/train/tensor_act_model_layers_45_self_attn_k_proj/norm":1286.3828999208806,"train/train/tensor_act_model_layers_25_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/mean":-6.426125764846802e-07,"train/train/tensor_act_model_layers_8_self_attn_o_proj/norm":280.0751191746162,"train/train/tensor_grad_model_norm_weight/max_abs":0.00616455078125,"train/train/tensor_act_model_layers_23_self_attn_q_proj/std":0.22534355010361123,"train/train/tensor_act_model_layers_93_self_attn_k_proj/max_abs":1.2265625,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/max_abs":3.9637088775634766e-06,"train/train/layer_model_layers_24/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_k_proj/norm":1268.817751398745,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/max_abs":0.0859375,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_18_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/norm":0.000123830166525104,"train/train/tensor_act_model_layers_68_self_attn_k_proj/std":0.22754749754955328,"train/train/tensor_act_model_layers_21_mlp_down_proj/max_abs":0.0859375,"train/train/tensor_act_model_layers_23_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_92_self_attn_k_proj/norm":1341.5382175013674,"train/train/tensor_act_model_layers_28_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_input_layernorm/max_abs":5.0625,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/mean":3.0386217986233532e-09,"train/train/tensor_act_model_layers_47_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_o_proj/max_abs":0.25,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/mean":9.71369445323944e-07,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_v_proj/max_abs":1.140625,"train/train/tensor_act_model_layers_74_mlp_up_proj/mean":0.0012197494506835938,"train/train/tensor_act_model_layers_49_self_attn_v_proj/max_abs":1.1796875,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp_up_proj/mean":0.0028839111328125,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/max_abs":0.080078125,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_53/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_25/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/max_abs":0.0791015625,"train/train/tensor_act_model_layers_65_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/std":5.041002072208395e-07,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_84/param/max_abs":1,"train/train/tensor_act_model_layers_62_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp_gate_proj/norm":1863.2837230185435,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_63/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_input_layernorm/max_abs":4.65625,"train/train/tensor_param_model_layers_45_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/max_abs":0.0031585693359375,"train/train/tensor_grad_model_layers_53_mlp_gate_proj_weight/mean":3.7695281207561493e-07,"train/train/tensor_act_model_layers_34_self_attn_o_proj/mean":0.0010347366333007812,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_v_proj/mean":-0.0083160400390625,"train/train/tensor_act_model_layers_50_mlp_up_proj/mean":0.016845703125,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/mean":0.00031280517578125,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_79/act/max_abs":4.6875,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/std":0.00010755670478537833,"train/train/tensor_act_model_layers_0_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_31/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn/mean":8.71419906616211e-05,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_mlp_gate_proj_weight/max_abs":0.0927734375,"train/train/tensor_act_model_layers_75_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_43/param/std":0.044248423120595175,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_38_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/max_abs":0.07666015625,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_65_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_mlp_gate_proj_weight/mean":0.00017547607421875,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_42/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/std":7.273534649857e-05,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_81_mlp_down_proj/mean":-0.0003380775451660156,"train/train/tensor_act_model_layers_68_self_attn_k_proj/max_abs":1,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/mean":-2.3283064365386963e-06,"train/train/tensor_grad_model_layers_50_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_27/grad/norm":0.2387881117889245,"train/train/layer_model_layers_84/act/norm":9305.367033384015,"train/train/layer__model_layers_76/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_66_mlp_gate_proj/norm":1881.3520617863567,"train/train/tensor_param_model_layers_29_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/mean":-0.00014019012451171875,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_mlp_gate_proj_weight/norm":0.024081830644970877,"train/train/tensor_act_model_layers_26/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/mean":5.459785461425781e-05,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/mean":0.00010967254638671875,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_51_mlp_down_proj/mean":-0.0003571510314941406,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_9_self_attn_k_proj/std":0.21704632567012894,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/max_abs":5.692243576049805e-06,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp/std":0.015426764818718871,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/max_abs":0.07763671875,"train/train/tensor_act_model_layers_78_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33/norm":1651.8343565400512,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/std":0.0004411300095396067,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/norm":0.00019419160337051373,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/mean":-0.00010442733764648438,"train/train/layer__model_layers_50/param/norm":17.928759848006777,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/mean":-1.5679688658565283e-09,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/std":0.020263671875,"train/train/tensor_param_model_layers_39_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/mean":3.039836883544922e-05,"train/train/layer__model_layers_44/param/mean":0.0016228948108119637,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/max_abs":0.07568359375,"train/train/tensor_act_model_layers_14_mlp_up_proj/norm":1874.434466697939,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_q_proj/max_abs":1.203125,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/max_abs":0.08984375,"train/train/tensor_param_model_layers_35_mlp_gate_proj_weight/mean":-1.823902130126953e-05,"train/train/tensor_act_model_layers_89_self_attn_o_proj/max_abs":0.2197265625,"train/train/layer__model_layers_91/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn/max_abs":0.2177734375,"train/train/tensor_act_model_layers_34_mlp_gate_proj/max_abs":1.203125,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/max_abs":0.004669189453125,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/max_abs":0.000804901123046875,"train/train/tensor_act_model_layers_42_mlp_gate_proj/norm":1838.7122802551496,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/std":7.869909947618694e-05,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_embed_tokens_weight/max_abs":0.10009765625,"train/train/layer_model_layers_36/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_11_mlp_gate_proj_weight/std":0.00020374516769731675,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/max_abs":0.09033203125,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/std":6.773860188032707e-05,"train/train/tensor_act_model_layers_46_self_attn/norm":277.4245531201896,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/std":0.0004149451263949252,"train/train/tensor_grad_model_layers_55_mlp_gate_proj_weight/norm":0.02644153990664566,"train/train/layer_model_layers_49/act/max_abs":4.71875,"train/train/tensor_act_model_layers_4_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer__model_layers_16/param/std":0.044265320463945906,"train/train/tensor_grad_model_layers_33_mlp_gate_proj_weight/norm":0.03950573274804584,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_63_input_layernorm/max_abs":4.75,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/norm":0.0006141840298374256,"train/train/tensor_act_model_layers_38_mlp/frac_near_user_limit":0,"train/train/layer__model_layers_65/param/frac_near_user_limit":0,"train/train/layer_model_layers_25/act/mean":-0.005812100001743862,"train/train/tensor_act_model_layers_61_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_input_layernorm/max_abs":4.96875,"train/train/tensor_act_model_layers_40_self_attn_q_proj/norm":1333.010639048308,"train/train/tensor_act_model_layers_68_self_attn_v_proj/norm":1330.021054543187,"train/train/layer_model_layers_20/grad/norm":0.32693107869064053,"train/train/tensor_act_model_layers_58_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_45_input_layernorm/std":1.0000038063299463,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/max_abs":0.078125,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/max_abs":0.011474609375,"train/train/tensor_act_model_layers_62_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_self_attn_v_proj/std":0.22437728158363912,"train/train/tensor_act_model_layers_0_mlp_gate_proj/norm":1844.8965747654754,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_gate_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/norm":2.53125,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/mean":3.309105522930622e-08,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/mean":4.559755325317383e-06,"train/train/tensor_act_model_layers_17_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_mlp_gate_proj_weight/mean":6.724148988723755e-07,"train/train/tensor_grad_model_layers_61_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/mean":-5.178662831895053e-09,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/mean":-1.2088567018508911e-06,"train/train/layer_model_layers_32/grad/max_abs":0.00848388671875,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/max_abs":0.0830078125,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/max_abs":0.0008544921875,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_61_self_attn_k_proj/norm":1305.1859363162316,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/mean":-0.002105712890625,"train/train/tensor_act_model_layers_37_input_layernorm/std":1.000004710851317,"train/train/layer_model_layers_48/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_q_proj/std":0.22388229173953034,"train/train/tensor_act_model_layers_9_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/std":3.3352498036239374e-05,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp/norm":85.25050333843588,"train/train/tensor_param_model_layers_3_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/std":0.0004763831915400522,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/max_abs":0.0830078125,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/std":0.0007948645304274841,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/mean":4.557892680168152e-06,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_83_self_attn_v_proj/mean":0.0102996826171875,"train/train/tensor_param_model_layers_59_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/std":0.0009561855962606254,"train/train/tensor_act_model_layers_91_self_attn_o_proj/std":0.04699854976164658,"train/train/tensor_act_model_layers_4_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_10/param/norm":17.91986556956413,"train/train/tensor_act_model_layers_45_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/std":0.000880161646187285,"train/train/tensor_act_model_layers_38_input_layernorm/norm":5792.578369145034,"train/train/layer__model_layers_19/param/mean":0.001645092659175117,"train/train/tensor_act_model_layers_44_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp/mean":0.0004982948303222656,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/norm":0.0021085581699095872,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_9_self_attn/norm":270.1541706032064,"train/train/tensor_act_model_layers_16_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/mean":3.3248215913772583e-07,"train/train/tensor_act_model_layers_72_mlp/std":0.015533686741760424,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_mlp_up_proj/norm":1869.2650023448984,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_mlp_gate_proj_weight/max_abs":0.09375,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn/mean":0.00325775146484375,"train/train/tensor_act_model_layers_78_self_attn/max_abs":0.263671875,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_53_mlp_gate_proj/std":0.22485557867402228,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/mean":-8.58306884765625e-05,"train/train/tensor_act_model_layers_15_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_92_self_attn/norm":294.7013685816614,"train/train/layer_model_layers_73/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_q_proj/mean":-0.00775146484375,"train/train/tensor_param_model_layers_4_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/max_abs":0.07958984375,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_7_self_attn_o_proj/std":0.04663394474949239,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/max_abs":0.0008087158203125,"train/train/layer_model_layers_76/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn/std":0.049135418958793954,"train/train/tensor_act_model_layers_28_self_attn_k_proj/std":0.2299831233243448,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_gate_proj_weight/max_abs":0.09423828125,"train/train/tensor_act_model_layers_78_mlp_up_proj/norm":1860.3500674457089,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/norm":0.02906311875371701,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/max_abs":0.000720977783203125,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_35_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/mean":-1.5122623153729364e-09,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_gate_proj/norm":1869.510208427854,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/max_abs":0.00101470947265625,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_mlp_gate_proj_weight/mean":-3.047134669031948e-07,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/mean":-2.6979250833392143e-08,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_post_attention_layernorm/mean":-0.03912353515625,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/norm":0.00013673092450491539,"train/train/tensor_act_model_layers_28_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/max_abs":0.00043487548828125,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/std":0.00041468394032173297,"train/train/tensor_grad_model_layers_51_mlp_gate_proj_weight/max_abs":0.0012054443359375,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/max_abs":2.1457672119140625e-05,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_mlp_gate_proj/max_abs":1.1953125,"train/train/tensor_act_model_layers_79/std":0.4458064868107821,"train/train/tensor_act_model_layers_30_mlp_gate_proj/max_abs":1.1328125,"train/train/tensor_act_model_layers_28_input_layernorm/std":1.0000026524031749,"train/train/tensor_act_model_layers_36_self_attn_k_proj/std":0.22144117626129795,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn_k_proj/norm":1295.7455737120406,"train/train/tensor_param_model_layers_84_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_41/mean":-0.00318145751953125,"train/train/tensor_act_model_layers_81_self_attn/mean":-0.0007419586181640625,"train/train/tensor_grad_model_layers_86_mlp_gate_proj_weight/norm":0.02475673515982771,"train/train/layer_model_layers_60/grad/mean":7.513728190370357e-08,"train/train/tensor_act_model_layers_59_post_attention_layernorm/mean":-0.007890701293945312,"train/train/tensor_param_model_layers_79_mlp_gate_proj_weight/max_abs":0.078125,"train/train/tensor_act_model_layers_90/norm":2783.8318082765386,"train/train/tensor_act_model_layers_78_mlp_down_proj/norm":90.95295758225211,"train/train/tensor_act_model_layers_44/norm":1880.1806159658768,"train/train/tensor_act_model_layers_55/norm":2124.5013548207085,"train/train/tensor_act_model_layers_47_self_attn/mean":-0.000396728515625,"train/train/tensor_act_model_layers_18_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_input_layernorm/mean":-0.004108428955078125,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_k_proj/norm":1312.1471450683791,"train/train/layer__model_layers_47/param/std":0.0442470481884605,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/max_abs":0.08349609375,"train/train/tensor_act_model_layers_20_self_attn_o_proj/norm":287.3581508952125,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/mean":-7.562339305877686e-07,"train/train/tensor_act_model_layers_41/frac_near_user_limit":0,"train/train/layer__model_layers_0/param/mean":0.0016311669312475624,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/mean":-9.113136911764741e-09,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/max_abs":0.00054168701171875,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_down_proj/mean":9.109079837799072e-05,"train/train/tensor_act_model_layers_70_input_layernorm/max_abs":4.65625,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/norm":0.00021830707070604087,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_mlp_up_proj/max_abs":1.2421875,"train/train/layer_model_layers_39/grad/norm":0.19609095274622582,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/mean":7.724761962890625e-05,"train/train/tensor_act_model_layers_34_self_attn_v_proj/max_abs":1.2109375,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/norm":0.0006505708680974347,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/max_abs":0.07470703125,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38/mean":-0.00543975830078125,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/std":4.276632492937861e-07,"train/train/layer__model_layers_20/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/max_abs":9.655952453613281e-06,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/std":8.983833164982443e-05,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_o_proj/norm":285.57978478466794,"train/train/layer_model_layers_38/grad/max_abs":0.005340576171875,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/std":0.0009017105943002525,"train/train/tensor_param_model_layers_46_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/std":1.3925909937155227e-06,"train/train/tensor_grad_model_layers_50_mlp_gate_proj_weight/norm":0.034977564773701356,"train/train/tensor_act_model_layers_56_self_attn_k_proj/mean":0.0005424022674560547,"train/train/tensor_grad_model_layers_57_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_o_proj/mean":0.0009365081787109375,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_93_mlp/std":0.015839508765506258,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/norm":0.0012621403091769628,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/max_abs":0.00112152099609375,"train/train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_mlp_down_proj/norm":89.32513311640379,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_mlp_gate_proj_weight/max_abs":0.0012969970703125,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/norm":0.03189663629540646,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/norm":0.022410626663662003,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_47/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/std":0.0010431106171570543,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/norm":0.12228106806424939,"train/train/layer__model_layers_9/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_o_proj/std":0.048524355118255535,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/norm":0.0008353076495211455,"train/train/tensor_act_model_layers_91_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/mean":-1.2405216693878174e-06,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/mean":0.00010919570922851562,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/mean":2.0422041416168213e-05,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_q_proj/norm":1384.2757273929692,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/mean":4.029273986816406e-05,"train/train/tensor_act_model_layers_2_post_attention_layernorm/std":1.0000107883297595,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/max_abs":0.08740234375,"train/train/layer_model_layers_53/grad/norm":0.16213303909601995,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/max_abs":3.844499588012695e-06,"train/train/tensor_param_model_layers_43_mlp_gate_proj_weight/max_abs":0.0869140625,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_90/param/norm":17.936363476625495,"train/train/tensor_act_model_layers_66_mlp_down_proj/norm":92.35709443455691,"train/train/tensor_act_model_layers_46_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp_gate_proj/max_abs":1.03125,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/norm":0.005611314649396437,"train/train/tensor_act_model_layers_32_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/std":0.0004891845424101867,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/mean":0.0002231597900390625,"train/train/tensor_act_model_layers_70/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/max_abs":0.0751953125,"train/train/tensor_act_model_layers_50_self_attn_v_proj/norm":1302.933070809079,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn/mean":0.0011615753173828125,"train/train/tensor_act_model_layers_57/max_abs":1.8046875,"train/train/tensor_grad_model_layers_7_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_mlp_gate_proj_weight/max_abs":0.00074005126953125,"train/train/tensor_act_model_layers_17_self_attn_k_proj/mean":0.0077362060546875,"train/train/layer__model_layers_46/param/std":0.044254615704079445,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/max_abs":0.000278472900390625,"train/train/tensor_param_model_layers_56_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_gate_proj_weight/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/std":4.632363563949774e-07,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/max_abs":0.0004863739013671875,"train/train/tensor_act_model_layers_75_mlp_down_proj/std":0.015839705856286513,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/max_abs":0.08154296875,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/norm":0.02985324269258262,"train/train/tensor_grad_model_layers_33_mlp_gate_proj_weight/max_abs":0.00150299072265625,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/mean":-1.7240643501281738e-05,"train/train/tensor_act_model_layers_60_self_attn_k_proj/max_abs":1.0546875,"train/train/tensor_act_model_layers_28_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/std":0.0007341951786587428,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/mean":7.82012939453125e-05,"train/train/tensor_act_model_layers_66_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn_v_proj/mean":-0.005096435546875,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/std":0.019775390625,"train/train/tensor_act_model_layers_71_self_attn/std":0.05023840358309806,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_40_mlp_down_proj/max_abs":0.08740234375,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_56_post_attention_layernorm/max_abs":4.96875,"train/train/tensor_act_model_layers_90_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/std":0,"train/train/layer__model_layers_32/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/norm":0.04900678374568166,"train/train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn/max_abs":0.2255859375,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/mean":6.293703336268663e-09,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/mean":-2.5391578674316406e-05,"train/train/layer_model_layers_24/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_mlp_up_proj/mean":0.00323486328125,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/norm":0.000172019954858537,"train/train/layer_model_layers_69/act/max_abs":4.625,"train/train/tensor_act_model_layers_91_self_attn_k_proj/max_abs":1.09375,"train/train/tensor_act_model_layers_18_mlp_gate_proj/max_abs":1.25,"train/train/tensor_act_model_layers_71_self_attn_q_proj/norm":1376.233534698124,"train/train/tensor_act_model/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_53_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/norm":0.001013626151349126,"train/train/tensor_act_model_layers_75_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/std":0.002397393366721398,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/mean":2.790329745039344e-09,"train/train/tensor_param_model_layers_34_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_24_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28/mean":-0.0089263916015625,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/norm":0.00012970146009719,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/norm":0.03425350392649445,"train/train/tensor_act_model_layers_11_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73/norm":2483.070651324325,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/mean":-1.382431946694851e-10,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_51_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_5/act/max_abs":5.34375,"train/train/tensor_grad_model_layers_14_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_21_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/norm":0.0001606842173797919,"train/train/layer_model_layers_27/act/std":0.41622853903707335,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/norm":0.0033648397803255486,"train/train/tensor_act_model_layers_45_self_attn/norm":291.7028413964522,"train/train/tensor_act_model_layers_55_self_attn_q_proj/std":0.22681352001306004,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/max_abs":5.0067901611328125e-06,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/mean":-2.276897430419922e-05,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_78_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_60_self_attn_q_proj/std":0.22144242804874842,"train/train/tensor_act_model_layers_30_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/std":9.810700512675386e-07,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/std":4.058592755088955e-05,"train/train/tensor_act_model_layers_58_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_input_layernorm/max_abs":4.46875,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/norm":0.00011983445273954844,"train/train/tensor_grad_model_layers_82_mlp_gate_proj_weight/mean":9.709037840366364e-08,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/max_abs":0.08447265625,"train/train/tensor_act_model_layers_45_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer_model_layers_67/act/std":0.4256649546288137,"train/train/tensor_act_model_layers_35_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_post_attention_layernorm/max_abs":4.65625,"train/train/layer_model_layers_56/grad/max_abs":0.00469970703125,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/mean":-3.7089193938300014e-09,"train/train/tensor_act_model_layers_72_mlp_up_proj/norm":1871.8646943352082,"train/train/tensor_act_model_layers_88/max_abs":2.234375,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_post_attention_layernorm/mean":-0.0537109375,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp_gate_proj/std":0.22852473429150155,"train/train/tensor_act_model_layers_20_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/mean":1.2369127944111824e-10,"train/train/tensor_act_model_layers_54/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp/mean":-5.638599395751953e-05,"train/train/tensor_act_model_layers_2_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_41/param/mean":0.00163464241206367,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/std":0.00046104819510450247,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/max_abs":0.07568359375,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/mean":0.000274658203125,"train/train/tensor_act_model_layers_93_self_attn_o_proj/max_abs":0.23046875,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/mean":-0.000244140625,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/norm":0.00020867714674476675,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/max_abs":0.000240325927734375,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/max_abs":0.0004119873046875,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/mean":3.962486516684294e-08,"train/train/tensor_act_model_layers_41_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_post_attention_layernorm/norm":5792.508056643305,"train/train/tensor_act_model_layers_85_input_layernorm/std":1.0000009019855067,"train/train/tensor_act_model_layers_43_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/norm":0.051699335424553135,"train/train/layer__model_layers_61/param/mean":0.001581987985024772,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/mean":0.00010395050048828125,"train/train/tensor_param_model_layers_54_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_25_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_64/param/std":0.04425009894970881,"train/train/tensor_act_model_layers_59_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_gate_proj_weight/mean":-2.8252601623535156e-05,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_v_proj/mean":0.0102691650390625,"train/train/tensor_act_model_layers_30_self_attn_v_proj/std":0.2236454137446559,"train/train/tensor_act_model_layers_72_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/max_abs":0.00408935546875,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/max_abs":0.08642578125,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_91_self_attn_k_proj/mean":0.00927734375,"train/train/tensor_act_model_layers_28_self_attn/norm":293.8350357628902,"train/train/tensor_act_model_layers_56_self_attn_o_proj/mean":0.0011844635009765625,"train/train/tensor_act_model_layers_34_self_attn/norm":281.18485254940197,"train/train/tensor_act_model_layers_86_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_15/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/std":7.902684175779291e-07,"train/train/tensor_act_model_layers_12_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/max_abs":0.0849609375,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/std":0.02001953125,"train/train/layer_model_layers_30/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp/std":0.016174759819406002,"train/train/tensor_param_model_layers_35_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/std":0.00043031738313330383,"train/train/layer__model_layers_54/param/max_abs":1,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/grad/mean":-5.572108593994574e-06,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/mean":4.260800778865814e-07,"train/train/tensor_param_model_layers_89_mlp_gate_proj_weight/mean":9.679794311523438e-05,"train/train/tensor_act_model_layers_17_post_attention_layernorm/std":1.0000015348184252,"train/train/layer__model_layers_79/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/mean":-6.998889148235321e-07,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/norm":2.546875,"train/train/layer_model_layers_29/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_78/act/max_abs":4.65625,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/norm":0.10255097819530877,"train/train/tensor_act_model_layers_36_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/std":3.3600007152623644e-05,"train/train/tensor_act_model_layers_30_self_attn_v_proj/norm":1295.547193071516,"train/train/tensor_act_model_layers_21_post_attention_layernorm/std":1.0000054016563438,"train/train/tensor_act_model_layers_35_post_attention_layernorm/std":1.00000472214355,"train/train/tensor_act_model_layers_51_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/mean":0.00013446807861328125,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/norm":3.65625,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_14_self_attn_q_proj/norm":1285.8474873470266,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/std":0.0001478308662848164,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/norm":3.640625,"train/train/layer_model_layers_36/grad/norm":0.224312632450448,"train/train/tensor_act_model_layers_80_self_attn_v_proj/max_abs":1.1484375,"train/train/tensor_act_model_layers_60_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn_v_proj/norm":1331.2254062186073,"train/train/tensor_grad_model_layers_55_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_mlp_gate_proj_weight/norm":0.024936054823643822,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/mean":2.2817403078079224e-08,"train/train/layer__model_layers_5/param/norm":17.930073865662823,"train/train/layer__model_layers_35/param/std":0.04425377859582899,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/max_abs":0.000225067138671875,"train/train/tensor_act_model_layers_53_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_3_mlp_up_proj/max_abs":1.171875,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/mean":0.0001068115234375,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/max_abs":0.08984375,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/std":0.02001953125,"train/train/layer_model_layers_87/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn_v_proj/max_abs":1.15625,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/std":0.0008742056387856362,"train/train/tensor_act_model_layers_78_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_58/act/std":0.4230629520095717,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn_o_proj/std":0.05023516609611818,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/max_abs":0.07666015625,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/mean":-0.0001621246337890625,"train/train/tensor_act_model_layers_84_post_attention_layernorm/norm":5792.5952148559845,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/mean":-3.259629011154175e-07,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/std":0.019775390625,"train/train/tensor_act_model_layers_17_self_attn_o_proj/mean":0.0007753372192382812,"train/train/tensor_act_model_layers_79/mean":0.0018548965454101562,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/max_abs":0.001556396484375,"train/train/tensor_param_model_layers_23_mlp_gate_proj_weight/mean":0.00019550323486328125,"train/train/layer__model_layers_7/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_k_proj/std":0.2204655168944775,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/mean":-3.695022314786911e-07,"train/train/layer_model_layers_40/act/norm":9080.626336306479,"train/train/tensor_act_model_layers_17_self_attn_k_proj/std":0.2270517928603412,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5/std":0.0932636285756076,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/std":5.427249685009222e-05,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_23_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/std":4.998814612927751e-07,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/mean":4.839897155761719e-05,"train/train/tensor_act_model_layers_93_mlp_gate_proj/max_abs":1.390625,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_gate_proj_weight/mean":9.441375732421875e-05,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/max_abs":0.083984375,"train/train/tensor_act_model_layers_64_self_attn/norm":293.0439414109723,"train/train/tensor_act_model_layers_22_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/mean":-3.130480763502419e-09,"train/train/layer_model_layers_17/grad/mean":1.9176672801212662e-07,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/norm":0.0002527673251797429,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_73/param/mean":0.0015479577312975331,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/norm":0.0020192832527036186,"train/train/tensor_act_model_layers_64_self_attn_q_proj/std":0.23389485206055727,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_33_mlp_down_proj/max_abs":0.0849609375,"train/train/tensor_act_model_layers_12_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/max_abs":0.0732421875,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/max_abs":0.078125,"train/train/tensor_act_model_layers_47_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_84_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/max_abs":0.0001373291015625,"train/train/tensor_act_model_layers_1_post_attention_layernorm/norm":5789.764160203947,"train/train/tensor_act_model_layers_52_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_1_mlp_gate_proj/std":0.2255862599067118,"train/train/layer_model_layers_90/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_66_mlp_gate_proj/max_abs":1.078125,"train/train/tensor_act_model_layers_65_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39/norm":1794.6997580992743,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/mean":-3.982335329055786e-06,"train/train/tensor_act_model_layers_80_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_mlp_gate_proj_weight/mean":-9.918212890625e-05,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/norm":2.546875,"train/train/layer_model_layers_24/grad/norm":0.27064584646749096,"train/train/layer__model_layers_64/param/mean":0.0016276676457683307,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_input_layernorm/mean":-0.0161285400390625,"train/train/tensor_act_model_layers_42_mlp_up_proj/max_abs":1.171875,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/mean":-1.68532133102417e-05,"train/train/tensor_act_model_layers_11_mlp_gate_proj/max_abs":1.125,"train/train/tensor_act_model_layers_57_self_attn_o_proj/std":0.048340807649847314,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/std":4.529887004392474e-07,"train/train/layer__model_layers_84/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/norm":0.03565999570892531,"train/train/tensor_act_model_layers_68_self_attn_o_proj/max_abs":0.23828125,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89/max_abs":2.21875,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_61/param/norm":17.939180833705173,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/max_abs":0.003082275390625,"train/train/tensor_act_model_layers_3_mlp/std":0.015244060172788388,"train/train/tensor_param_model_layers_36_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_49_mlp_gate_proj/norm":1864.9515616870806,"train/train/layer__model_layers_13/param/max_abs":1,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/mean":4.547182470560074e-07,"train/train/tensor_act_model_layers_73_mlp_gate_proj/norm":1860.9618835417068,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/norm":0.11302181659511025,"train/train/tensor_act_model_layers_54_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_gate_proj/std":0.22436746600331126,"train/train/tensor_act_model_layers_37_input_layernorm/norm":5792.5745849633595,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/max_abs":0.095703125,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/norm":0.15403242022610983,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/mean":1.3387762010097504e-08,"train/train/tensor_act_model_layers_21_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/norm":0.13720703125,"train/train/layer_model_layers_35/grad/std":0.00028145204151468683,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/max_abs":6.109476089477539e-06,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/mean":4.019588232040405e-06,"train/train/tensor_act_model_layers_43_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/mean":-1.3870885595679283e-07,"train/train/tensor_act_model_layers_2/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_15/act/max_abs":5.03125,"train/train/layer_model_layers_12/grad/norm":0.39985093816726286,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/mean":-0.000152587890625,"train/train/layer__model_layers_74/param/norm":17.926533287368336,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_35/max_abs":1.515625,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_k_proj/max_abs":1.1015625,"train/train/tensor_act_model_layers_21_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_4/act/mean":-0.007797377450125558,"train/train/tensor_act_model_layers_12_mlp_up_proj/mean":-0.0001850724220275879,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_22/act/norm":8991.563544843373,"train/train/tensor_act_model_layers_75_self_attn_q_proj/mean":-0.0103607177734375,"train/train/tensor_act_model_layers_71_self_attn_o_proj/std":0.05023840358309806,"train/train/tensor_act_model_layers_47_mlp/max_abs":0.08935546875,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/std":0.00040294077185621345,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/mean":8.0108642578125e-05,"train/train/tensor_act_model_layers_79_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_v_proj/norm":1310.3502612214386,"train/train/tensor_grad_model_layers_14_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/max_abs":0.004180908203125,"train/train/tensor_act_model_layers_7/std":0.11780080050399253,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/max_abs":0.083984375,"train/train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_input_layernorm/max_abs":4.75,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_66/param/mean":0.0014650639431338973,"train/train/tensor_act_model_layers_59/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/mean":3.3778633223846555e-09,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/norm":0.0001449177082300113,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/mean":-0.0002765655517578125,"train/train/tensor_param_model_layers_57_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_85_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/max_abs":0.00052642822265625,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/mean":6.771087646484375e-05,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/norm":0.0008177633534821707,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_29_mlp_gate_proj/mean":0.0018062591552734375,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_45_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn/mean":-0.0006604194641113281,"train/train/tensor_act_model_layers_36_mlp/max_abs":0.09228515625,"train/train/tensor_act_model_layers_67/mean":0.0013890266418457031,"train/train/layer_model_layers_36/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_67_post_attention_layernorm/std":1.000003877637695,"train/train/layer__model_layers_1/param/norm":17.925593554939123,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_input_layernorm/norm":5792.589721682295,"train/train/layer_model_layers_39/grad/max_abs":0.0047607421875,"train/train/layer_model_layers_61/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn/max_abs":0.24609375,"train/train/tensor_grad_model_layers_40_mlp_gate_proj_weight/std":0.0001088896057479551,"train/train/tensor_act_model_layers_68_mlp_gate_proj/std":0.22705167042097635,"train/train/tensor_act_model_layers_72/std":0.4257887024036555,"train/train/tensor_act_model_layers_80_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_9/grad/mean":-2.190551105610643e-06,"train/train/tensor_act_model_layers_2_self_attn/max_abs":0.251953125,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/max_abs":0.0008392333984375,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/std":8.846368865745465e-05,"train/train/tensor_act_model_layers_27_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_74/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn/norm":266.3765409383034,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/mean":9.521842002868652e-06,"train/train/tensor_act_model_layers_41_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/max_abs":0.00142669677734375,"train/train/tensor_grad_model_layers_76_mlp_gate_proj_weight/mean":-1.5331897884607315e-07,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_o_proj/norm":266.3765409383034,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_mlp_down_proj/mean":0.0003819465637207031,"train/train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/std":0.00010441647909397613,"_wandb":{"runtime":85},"train/train/tensor_grad_model_layers_27_input_layernorm_weight/norm":0.0033646667879662345,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_post_attention_layernorm/std":0.9990279793286491,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/std":0.0003705882845619091,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/std":5.772886581384299e-05,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_18_mlp_gate_proj_weight/max_abs":0.08447265625,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/std":6.679896200498328e-05,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_65_self_attn/std":0.051026963392844904,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/mean":-8.614733815193176e-07,"train/train/tensor_param_model_layers_13_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/max_abs":5.27501106262207e-06,"train/train/tensor_act_model_layers_15_self_attn_v_proj/std":0.226564979102901,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/norm":0.49246740707755615,"train/train/tensor_act_model_layers_26_mlp_up_proj/mean":-0.00519561767578125,"train/train/tensor_act_model_layers_17_self_attn/max_abs":0.23046875,"train/train/tensor_act_model_layers_12_post_attention_layernorm/mean":-0.050537109375,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/mean":9.202957153320312e-05,"train/train/tensor_act_model_layers_88_self_attn_v_proj/max_abs":1.0625,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/mean":0.0001201629638671875,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_74/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_q_proj/std":0.22387938461730497,"train/train/tensor_act_model_layers_70_self_attn_v_proj/norm":1308.635240102471,"train/train/layer__model_layers_69/param/mean":0.001528866391472065,"train/train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_q_proj/norm":1287.2899662268906,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/mean":-8.509960025548935e-07,"train/train/tensor_act_model_layers_66_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/norm":0.12515368550658076,"train/train/tensor_param_model_layers_5_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_v_proj/norm":1294.5641514188778,"train/train/tensor_param_model_layers_19_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_down_proj/max_abs":0.07958984375,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/mean":-5.626678466796875e-05,"train/train/tensor_act_model_layers_22_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/max_abs":0.08154296875,"train/train/tensor_act_model_layers_35_mlp/norm":91.51150192762687,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/mean":7.390975952148438e-05,"train/train/tensor_act_model_layers_28_mlp/std":0.015443274157754145,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_77_input_layernorm/std":1.0000009956997518,"train/train/tensor_act_model_layers_90_self_attn_v_proj/max_abs":1.078125,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/mean":-9.822845458984375e-05,"train/train/tensor_act_model_layers_87_mlp_up_proj/max_abs":1.0703125,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_31_self_attn_o_proj/norm":274.9124768963415,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/std":0.0019097731229793442,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_80/grad/max_abs":0.005462646484375,"train/train/tensor_act_model_layers_1_mlp_down_proj/max_abs":0.09423828125,"train/train/tensor_param_model_layers_79_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/norm":0.00012458049491023695,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/mean":6.407499313354492e-06,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/mean":5.507899913936853e-09,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/max_abs":0.00433349609375,"train/train/layer__model_layers_12/param/max_abs":1,"train/train/tensor_act_model_layers_35_mlp/mean":0.000400543212890625,"train/train/tensor_act_model_layers_21_mlp_gate_proj/std":0.22485471602075388,"train/train/tensor_act_model_layers_51_self_attn_k_proj/norm":1329.6233127450298,"train/train/tensor_act_model_layers_26_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/mean":3.769993782043457e-06,"train/train/tensor_act_model_layers_19/mean":-0.012298583984375,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/max_abs":0.08056640625,"train/train/tensor_grad_model_layers_31_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_59_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/norm":3.625,"train/train/layer_model_layers_26/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/mean":1.1525116860866547e-08,"train/learning_rate":0.001,"train/train/tensor_act_model_layers_82_self_attn_o_proj/norm":289.35159843022666,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/mean":-1.5742843970656395e-06,"train/train/tensor_act_model_layers_11_mlp_gate_proj/mean":-0.005062103271484375,"train/train/tensor_act_model_layers_64_self_attn/mean":-0.00015676021575927734,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_29/grad/mean":-8.979511509726461e-07,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/std":0.0005745461958535487,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/max_abs":0.000835418701171875,"train/train/tensor_act_model_layers_44_self_attn_v_proj/norm":1344.4184832702033,"train/train/tensor_param_model_layers_44_mlp_gate_proj_weight/norm":3.671875,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_62_mlp_down_proj/std":0.015503561349066197,"train/train/tensor_act_model_layers_0_input_layernorm/std":1.0000000828076703,"train/train/tensor_act_model_layers_69_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_input_layernorm_weight/mean":1,"train/train/layer__model_layers_2/param/max_abs":1,"train/train/tensor_act_model_layers_78_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_24/act/mean":-0.0075487324169703895,"train/train/tensor_act_model_layers_11_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_input_layernorm/mean":-0.0556640625,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_49/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_6_self_attn/mean":-0.003711700439453125,"train/train/tensor_act_model_layers_59_self_attn_q_proj/norm":1309.4571642456135,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_input_layernorm/max_abs":4.53125,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_20_mlp_gate_proj_weight/mean":7.743947207927704e-07,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/mean":-3.981590270996094e-05,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_50_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_2/grad/max_abs":0.0308837890625,"train/train/tensor_act_model_layers_60_self_attn_o_proj/max_abs":0.234375,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/norm":0.2816803077539628,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/mean":7.180497050285339e-07,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/mean":8.378992788493633e-08,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_15_input_layernorm/mean":-0.05352783203125,"train/train/tensor_act_model_layers_49_self_attn_k_proj/std":0.2285265307022368,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/norm":0.03901383229247762,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/norm":0.12258412372603132,"train/train/tensor_act_model_layers_27_mlp_up_proj/norm":1829.8186927591262,"train/train/tensor_act_model_layers_92_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_21_mlp_gate_proj_weight/norm":0.04793339174812659,"train/train/tensor_param_model_layers_6_mlp_gate_proj_weight/max_abs":0.08544921875,"train/train/tensor_act_model_layers_23_mlp_gate_proj/mean":0.00896453857421875,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn/std":0.048524355118255535,"train/train/tensor_act_model_layers_57_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_56/param/norm":17.931040594368053,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/std":8.038434747572577e-05,"train/train/tensor_act_model_layers_50_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_input_layernorm/std":1.0000078049984131,"train/train/tensor_act_model_layers_81_self_attn_o_proj/mean":-0.0007419586181640625,"train/train/tensor_act_model_layers_91_self_attn_o_proj/max_abs":0.22265625,"train/train/layer_model_layers_14/act/norm":8970.48383415011,"train/train/layer_model_layers_0/grad/norm":1.2796610274803681,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_70/max_abs":1.953125,"train/train/layer__model_layers_12/param/norm":17.93680584458253,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/max_abs":0.002227783203125,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/norm":2.546875,"train/train/layer_model_layers_3/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/max_abs":4.410743713378906e-06,"train/train/tensor_act_model_layers_84_mlp/mean":0.0003819465637207031,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/std":0.00043673196921773335,"train/train/tensor_act_model_layers_34_self_attn_k_proj/mean":0.0094146728515625,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/mean":1.823902130126953e-05,"train/train/tensor_act_model_layers_78_post_attention_layernorm/std":1.0000018643805462,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/max_abs":0.005615234375,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64/std":0.39844662650517804,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/mean":1.2316741049289703e-07,"train/train/layer_model_layers_39/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/max_abs":0.0771484375,"train/train/tensor_param_model_layers_14_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn/mean":0.0010347366333007812,"train/train/tensor_grad_model_layers_46_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/norm":0.003073282625812958,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/std":6.816070100700179e-07,"train/train/tensor_act_model_layers_87_input_layernorm/max_abs":4.6875,"train/train/tensor_act_model_layers_54_self_attn_v_proj/mean":-0.005340576171875,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_mlp_gate_proj_weight/mean":-1.7578713595867157e-07,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/norm":0.0016631511363534152,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_post_attention_layernorm/norm":5792.545532226782,"train/train/tensor_act_model_layers_49_mlp_down_proj/norm":90.89383213770765,"train/train/tensor_grad_model_layers_38_mlp_gate_proj_weight/mean":1.942971721291542e-07,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/norm":0.00012044197757850717,"train/train/layer_model_layers_53/grad/max_abs":0.00421142578125,"train/train/tensor_grad_model_layers_77_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/max_abs":0.0908203125,"train/train/layer_model_layers_70/act/std":0.42618771391913784,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/mean":-2.3051143216434866e-09,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/std":9.696276797853292e-05,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/max_abs":0.000133514404296875,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/mean":-7.152557373046875e-05,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/mean":2.574920654296875e-05,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/max_abs":0.004364013671875,"train/train/tensor_act_model_layers_64_mlp_up_proj/std":0.22558819817544573,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/mean":4.863739013671875e-05,"train/train/tensor_act_model_layers_90_self_attn_k_proj/norm":1314.1079175378025,"train/train/tensor_act_model_layers_11_input_layernorm/norm":5792.479125979335,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/mean":6.914138793945312e-05,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/std":7.276206266786539e-07,"train/train/tensor_act_model_layers_1_self_attn_o_proj/max_abs":0.224609375,"train/train/tensor_act_model_layers_13_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_mlp_down_proj/norm":88.27401653108261,"train/train/tensor_act_model_layers_62/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_52_self_attn_k_proj/norm":1287.6568218637537,"train/train/tensor_act_model_layers_82_mlp_down_proj/max_abs":0.08935546875,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/max_abs":0.08837890625,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/std":0.00018049702435655982,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/mean":-9.298324584960938e-05,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/norm":0.0001514361545510331,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_71_mlp_gate_proj_weight/mean":-0.00020313262939453125,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_36_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_k_proj/max_abs":1.0703125,"train/train/tensor_act_model_layers_9_mlp/std":0.015259784904408845,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/mean":2.091837814077735e-09,"train/train/tensor_param_model_layers_25_input_layernorm_weight/mean":1,"train/train/layer__model_layers_51/param/std":0.0442601203972613,"train/train/tensor_act_model_layers_91_self_attn/max_abs":0.22265625,"train/train/tensor_param_model_layers_64_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_92_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_66_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81/std":0.4541051259945696,"train/train/tensor_act_model_layers_1_mlp_up_proj/std":0.22460940169251326,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/max_abs":0.0810546875,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/std":6.311379639734755e-07,"train/train/layer_model_layers_21/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/max_abs":0.005584716796875,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/norm":0.03284803010907213,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/norm":0.00035649820565655793,"train/train/tensor_param_model_layers_41_mlp_gate_proj_weight/mean":1.0907649993896484e-05,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_8_self_attn_o_proj/std":0.048280399280631305,"train/train/tensor_act_model_layers_9_self_attn_k_proj/max_abs":1.1328125,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/max_abs":0.0018310546875,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/mean":8.831193554215133e-09,"train/train/tensor_act_model_layers_43_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_63/grad/mean":4.496940450856169e-07,"train/train/tensor_act_model_layers_50_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp_gate_proj/std":0.22583134815142472,"train/train/tensor_param_model_layers_73_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_38_mlp_gate_proj_weight/norm":0.03413104274192964,"train/train/tensor_act_model_layers_36_input_layernorm/norm":5792.574707033853,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/mean":9.473878890275955e-07,"train/train/tensor_act_model_layers_7_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp/norm":92.32689033386708,"train/train/tensor_grad_model_layers_93_mlp_gate_proj_weight/norm":0.026395993921463322,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_15_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model/std":1.0000013913949586,"train/train/layer_model_layers_25/act/std":0.41559574880815153,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/mean":-4.500150680541992e-06,"train/train/layer_model_layers_17/grad/norm":0.34170092016915404,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/norm":3.59375,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/std":0.00016943066962328101,"train/train/tensor_grad_model_layers_85_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/norm":0.0002914594725429164,"train/train/layer_model_layers_74/act/mean":0.0016328947884695871,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/norm":2.59375,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/mean":-1.1622905731201172e-05,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/mean":-3.663444658741355e-09,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/max_abs":0.07958984375,"train/train/tensor_act_model_layers_15_mlp/max_abs":0.0830078125,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/norm":0.03015581476559605,"train/train/tensor_act_model_layers_46_post_attention_layernorm/max_abs":4.75,"train/train/tensor_act_model_layers_9_input_layernorm/std":1.0000040531076309,"train/train/tensor_act_model_layers_71_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/mean":0.0001621246337890625,"train/train/tensor_act_model_layers_7_self_attn_v_proj/max_abs":1.140625,"train/train/tensor_act_model_layers_32/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/std":4.607176004827706e-07,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/std":0.000267205947387365,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_gate_proj/mean":0.0056610107421875,"train/train/layer_model_layers_54/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/mean":-0.0003108978271484375,"train/train/tensor_act_model_layers_22_post_attention_layernorm/max_abs":4.6875,"train/train/layer_model_layers_21/act/std":0.4149018550928019,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/mean":0.000270843505859375,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_71_self_attn_v_proj/max_abs":1.1015625,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_76_self_attn_k_proj/max_abs":1.1015625,"train/train/layer__model_layers_9/param/norm":17.930986132164485,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_input_layernorm/mean":-0.006927490234375,"train/train/tensor_act_model_layers_5_mlp_up_proj/max_abs":1.203125,"train/train/tensor_act_model_layers_67_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_14/grad/std":0.0004695746641912246,"train/train/tensor_param_model_layers_54_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_mlp_gate_proj_weight/norm":0.05460951558150941,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/mean":0.00012063980102539062,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/mean":-2.1338462829589844e-05,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/mean":4.223547875881195e-07,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_39_self_attn_k_proj/std":0.23682394688561187,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/norm":0.10322414602830438,"train/train/tensor_param_model_layers_0_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_73_input_layernorm/max_abs":4.53125,"train/train/tensor_act_model_layers_58_self_attn/mean":-0.0012836456298828125,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/mean":3.971217665821314e-07,"train/train/layer_model_layers_31/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp_down_proj/mean":8.672475814819336e-05,"train/train/tensor_act_model_layers_13_mlp_gate_proj/norm":1852.3537471336042,"train/train/tensor_act_model_layers_13_mlp_up_proj/std":0.22607748536346253,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/norm":0.0019402564318908295,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/mean":2.280576154589653e-07,"train/train/tensor_act_model_layers_87/max_abs":2.21875,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/mean":8.96453857421875e-05,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/norm":0.033886551650457064,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_1_post_attention_layernorm/mean":-0.054931640625,"train/train/tensor_act_model_layers_20_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/max_abs":0.087890625,"train/train/tensor_grad_model_layers_21_mlp_gate_proj_weight/mean":7.674098014831543e-07,"train/train/tensor_act_model_layers_38/max_abs":1.546875,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/std":9.197181904748575e-05,"train/train/tensor_act_model_layers_91_post_attention_layernorm/mean":0.0133819580078125,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn_k_proj/norm":1292.154606981994,"train/train/tensor_grad_model_layers_6_mlp_gate_proj_weight/std":0.0002677048731918853,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_mlp_gate_proj_weight/std":0.0001167686986057338,"train/train/tensor_act_model_layers_24_mlp_down_proj/max_abs":0.083984375,"train/train/tensor_act_model_layers_59_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/mean":8.940696716308594e-06,"train/train/tensor_act_model_layers_80_self_attn_k_proj/std":0.22876945635891174,"train/train/tensor_act_model_layers_47_mlp/mean":5.5149197578430176e-05,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/max_abs":0.08740234375,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/std":9.803098070231723e-05,"train/train/layer__model_layers_75/param/max_abs":1,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/mean":1.760781742632389e-09,"train/train/tensor_act_model_layers_0_self_attn_q_proj/std":0.22460939847742611,"train/train/layer_model_layers_86/act/norm":9311.126337000806,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_42_self_attn/norm":285.5074063439483,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/norm":0.0001142373360398568,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/std":1.374675538575927e-06,"train/train/tensor_grad_model_layers_43_mlp_gate_proj_weight/std":9.371399853017666e-05,"train/train/tensor_param_model_layers_5_mlp_gate_proj_weight/mean":-8.392333984375e-05,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_91_mlp_gate_proj_weight/max_abs":0.0927734375,"train/train/tensor_act_model_layers_66/max_abs":1.921875,"train/train/tensor_act_model_layers_85/mean":0.0043792724609375,"train/train/layer_model_layers_87/grad/std":0.00016039064951064798,"train/train/tensor_act_model_layers_76_mlp_gate_proj/std":0.22021585329309676,"train/train/tensor_act_model_layers_15_self_attn_k_proj/mean":0.0013468265533447266,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_mlp_up_proj/mean":0.0120391845703125,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/mean":-0.0001201629638671875,"train/train/tensor_act_model_layers_56_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/max_abs":0.083984375,"train/train/tensor_act_model_layers_74_self_attn/std":0.04980673846918754,"train/train/tensor_act_model_layers_38_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn_v_proj/std":0.22242161869694876,"train/train/tensor_act_model_layers_16_self_attn/max_abs":0.263671875,"train/train/tensor_act_model_layers_77_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/mean":-0.0001087188720703125,"train/train/tensor_act_model_layers_78_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer_model_layers_88/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/max_abs":0.08251953125,"train/train/tensor_param_model_layers_34_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_11_mlp_up_proj/mean":-0.004669189453125,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_q_proj/mean":-0.01556396484375,"train/train/tensor_param_model_layers_38_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_8_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/std":0.23047850879116236,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/max_abs":0.0030364990234375,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/std":0.00012502290412452446,"train/train/tensor_act_model_layers_28/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/max_abs":0.0849609375,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/max_abs":1.1324882507324219e-05,"train/train/tensor_act_model_layers_17_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_mlp_gate_proj_weight/max_abs":0.00168609619140625,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/max_abs":0.00015544891357421875,"train/train/tensor_param_model_layers_14_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/mean":-8.67992639541626e-07,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/mean":0.00023174285888671875,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/norm":0.11241661796446231,"train/train/tensor_act_model_layers_87/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_76_self_attn_o_proj/norm":281.4429203156817,"train/train/tensor_act_model_layers_29_self_attn_k_proj/std":0.218509775664754,"train/train/tensor_act_model_layers_53_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_v_proj/norm":1330.5063117753402,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/mean":-4.237517714500427e-07,"train/train/tensor_act_model_layers_67_mlp_up_proj/norm":1857.6507563308692,"train/train/tensor_act_model_layers_66_mlp/mean":-0.00038814544677734375,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/mean":-2.0349398255348206e-06,"train/train/layer_model_layers_2/act/std":0.4107406743128482,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_down_proj/max_abs":0.0830078125,"train/train/tensor_act_model_layers_56_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_14/act/std":0.4143048254965115,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/max_abs":0.08251953125,"train/train/tensor_act_model_layers_14_mlp_up_proj/std":0.22876421960886836,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/mean":-6.723403930664062e-05,"train/train/tensor_act_model_layers_6_self_attn_o_proj/max_abs":0.2294921875,"train/train/tensor_act_model_layers_15_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_mlp_gate_proj/mean":-0.00777435302734375,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/mean":1.633167266845703e-05,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/norm":0.0440193400468972,"train/train/layer_model_layers_67/grad/std":0.00020526907692620353,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_gate_proj/std":0.2260749449149245,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/mean":5.936622619628906e-05,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/max_abs":0.08349609375,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/max_abs":0.0012969970703125,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/norm":9.238989118530899e-05,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/max_abs":0.087890625,"train/train/tensor_act_model_layers_15_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_mlp_down_proj/std":0.015429028782923203,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/mean":-8.20159912109375e-05,"train/train/tensor_act_model_layers_20/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/mean":-2.204387783422135e-09,"train/train/tensor_act_model_layers_66_self_attn_q_proj/max_abs":1.1328125,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/mean":8.866190910339355e-06,"train/train/tensor_act_model_layers_22/mean":-0.0138092041015625,"train/train/tensor_act_model_layers_8_post_attention_layernorm/norm":5792.4270019621235,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_49_post_attention_layernorm/std":1.0000033162013588,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_k_proj/std":0.22851850011002137,"train/train/tensor_grad_model_layers_76_mlp_gate_proj_weight/norm":0.025927300458535092,"train/train/tensor_act_model_layers_53_self_attn_v_proj/norm":1275.4821704420458,"train/train/tensor_act_model_layers_70_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_29/act/max_abs":4.78125,"train/train/tensor_param_model_layers_18_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_embed_tokens/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp/norm":86.85686982149316,"train/train/tensor_act_model_layers_65_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/mean":-2.7064234018325806e-06,"train/train/tensor_act_model_layers_76_post_attention_layernorm/norm":5792.598999023915,"train/train/layer_model_layers_30/act/std":0.41779008470880125,"train/train/tensor_act_model_layers_81_mlp_gate_proj/mean":-0.0051422119140625,"train/train/tensor_act_model_layers_1_self_attn_q_proj/mean":-0.0096435546875,"train/train/tensor_act_model_layers_32_mlp_up_proj/norm":1830.980438357742,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/mean":3.147125244140625e-05,"train/train/layer_model_layers_59/act/norm":9184.150116639814,"train/train/tensor_act_model_layers_62_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_17_input_layernorm/std":0.9980507475447503,"train/train/tensor_param_model_layers_74_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_88_mlp_up_proj/mean":0.004489898681640625,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93/max_abs":2.265625,"train/train/tensor_param_model_layers_56_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/std":4.3829353643110526e-05,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/mean":-7.180497050285339e-07,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_mlp/max_abs":0.08203125,"train/train/tensor_act_model_layers_67_mlp_gate_proj/std":0.2275464601347671,"train/train/tensor_act_model_layers_88_self_attn_v_proj/std":0.22730114320871952,"train/train/layer__model_layers_13/param/std":0.044257906370362386,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_q_proj/std":0.2238871635513127,"train/train/layer_model_layers_32/grad/std":0.0002727127961397699,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_14_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_q_proj/mean":0.0088653564453125,"train/train/tensor_act_model_layers_40_mlp/norm":88.92811747843332,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_gate_proj/mean":0.0005990862846374512,"train/train/tensor_act_model_layers_7_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6/max_abs":0.65625,"train/train/tensor_act_model_layers_68_input_layernorm/max_abs":4.40625,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/std":0.00013729024417256634,"train/train/tensor_act_model_layers_28_mlp_down_proj/max_abs":0.08154296875,"train/train/tensor_act_model_layers_2_self_attn_o_proj/std":0.031037133401281308,"train/train/tensor_grad_model_layers_43_mlp_gate_proj_weight/max_abs":0.00131988525390625,"train/train/tensor_act_model_layers_64_self_attn_o_proj/std":0.05054102582989184,"train/train/tensor_act_model_layers_38_self_attn_k_proj/max_abs":1.0546875,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_47_self_attn_v_proj/norm":1349.4213837315915,"train/train/tensor_act_model_layers_7_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_post_attention_layernorm/mean":-0.0567626953125,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/max_abs":0.08349609375,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp_down_proj/norm":93.35886087081747,"train/train/layer_model_layers_25/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_v_proj/max_abs":1.1328125,"train/train/layer__model_layers_71/param/max_abs":1,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_15_mlp_up_proj/std":0.22876557105645887,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/mean":-6.0498714447021484e-06,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/std":9.722995033001152e-07,"train/train/tensor_param_model_layers_93_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/max_abs":0.0072021484375,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_up_proj/mean":0.002899169921875,"train/train/tensor_act_model_layers_44_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_v_proj/std":0.22656513427875957,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/std":0,"train/train/layer__model_layers_14/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn_q_proj/max_abs":1.0625,"train/train/tensor_act_model_layers_77/mean":0.0015568733215332031,"train/train/tensor_act_model_layers_23/std":0.23755354254220826,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_mlp_gate_proj_weight/max_abs":0.000820159912109375,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_33_mlp_up_proj/max_abs":1.0546875,"train/train/tensor_act_model_layers_83_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_41/grad/norm":0.19104937658210053,"train/train/tensor_act_model_layers_4_input_layernorm/std":1.000024234582379,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/mean":0.00010251998901367188,"train/train/layer_model_layers_43/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_input_layernorm/norm":5792.515014660697,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/max_abs":0.00051116943359375,"train/train/tensor_act_model_layers_11_mlp_gate_proj/std":0.22705299990990252,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/max_abs":0.00084686279296875,"train/train/tensor_act_model_layers_7_self_attn_q_proj/mean":-0.003955841064453125,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/std":1.2657022469658095e-06,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_71_self_attn_q_proj/std":0.2375493555305699,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp/std":0.01570165041606858,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/mean":2.176966518163681e-07,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_42_input_layernorm/max_abs":4.90625,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/max_abs":0.001129150390625,"train/train/tensor_act_model_layers_22_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_q_proj/norm":1349.0568330381107,"train/train/tensor_act_model_layers_59_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/std":0.000368275314596565,"train/train/tensor_act_model_layers_13_mlp_down_proj/std":0.015810022945638966,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/std":8.740090028245747e-05,"train/train/layer_model_layers_54/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_post_attention_layernorm/mean":-0.0007305145263671875,"train/train/tensor_act_model_layers_1_input_layernorm/max_abs":4.75,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/std":0.00044149170571267186,"train/train/tensor_act_model_layers_84_input_layernorm/max_abs":4.71875,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/mean":4.792213439941406e-05,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn_v_proj/mean":0.00724029541015625,"train/train/tensor_act_model_layers_93_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/max_abs":0.08447265625,"train/train/tensor_act_model_layers_91_mlp/mean":-0.00043582916259765625,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/max_abs":0.004425048828125,"train/train/tensor_act_model_layers_35_self_attn_q_proj/mean":-0.00281524658203125,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_mlp_gate_proj_weight/max_abs":0.0045166015625,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/norm":0.0017987865084391195,"train/train/tensor_act_model_layers_12_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_q_proj/mean":-0.00690460205078125,"train/train/tensor_param_model_layers_82_mlp_gate_proj_weight/mean":-0.00025177001953125,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/norm":0.031778456862449254,"train/train/tensor_act_model_layers_40_mlp_up_proj/max_abs":1.1875,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/std":6.586147151881431e-05,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/max_abs":3.5315752029418945e-06,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/max_abs":0.0888671875,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_40/act/std":0.4189930387075696,"train/train/tensor_act_model_layers_11_input_layernorm/mean":-0.058837890625,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_mlp_gate_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_24_self_attn_k_proj/norm":1294.8468515315315,"train/train/tensor_act_model_layers_18_mlp_up_proj/max_abs":1.140625,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_o_proj/max_abs":0.212890625,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_41_mlp/max_abs":0.08251953125,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_13_mlp_down_proj/norm":91.64267406011611,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/max_abs":0.003204345703125,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/norm":2.515625,"train/train/tensor_act_model_layers_81/max_abs":2.171875,"train/train/tensor_grad_model_layers_77_mlp_gate_proj_weight/std":7.521494580654313e-05,"train/train/tensor_act_model_layers_6/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/std":1.363445262147483e-06,"train/train/tensor_param_model_layers_52_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/std":0.0003979268438023989,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_60_input_layernorm/mean":-0.008029937744140625,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/max_abs":0.00531005859375,"train/train/tensor_act_model_layers_63_mlp_up_proj/mean":0.00276947021484375,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn/norm":271.9793402297586,"train/train/tensor_grad_model_layers_30_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/mean":-6.723403930664062e-05,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/max_abs":0.000213623046875,"train/train/tensor_act_model_layers_69_self_attn_v_proj/mean":-0.0009199380874633789,"train/train/tensor_param_model_layers_88_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_47_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn/max_abs":0.251953125,"train/train/tensor_act_model_layers_88_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_71/grad/std":0.00019181824085873813,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/norm":0.03132358470560001,"train/train/layer_model_layers_76/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_mlp_gate_proj_weight/mean":3.7159770727157593e-07,"train/train/layer_model_layers_71/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp/max_abs":0.09423828125,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/mean":-1.9490718841552734e-05,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_87_self_attn_v_proj/max_abs":1.171875,"train/train/global/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/std":0.00024620311627201774,"train/train/tensor_act_model_layers_24_self_attn_v_proj/max_abs":1.1015625,"train/train/layer__model_layers_38/param/mean":0.0015898182314011311,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/max_abs":7.152557373046875e-06,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/norm":0.0007172873571958072,"train/train/tensor_param_model_layers_73_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87/mean":0.005157470703125,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/std":0.00038707507001861305,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_o_proj/mean":-0.0006821155548095703,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_20_mlp_down_proj/std":0.015930739665679237,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_48_self_attn_q_proj/max_abs":1.0390625,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/max_abs":0.0859375,"train/train/tensor_act_model_layers_52_post_attention_layernorm/max_abs":4.84375,"train/train/tensor_act_model_layers_77_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/norm":0.10615907031858833,"train/train/layer__model_layers_41/param/norm":17.932313601254858,"train/train/tensor_act_model_layers_23_mlp_up_proj/std":0.22876199789393192,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_gate_proj/std":0.22705240774597543,"train/train/tensor_act_model_layers_59_self_attn_k_proj/norm":1293.2388611167924,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/mean":6.246846169233322e-07,"train/train/tensor_act_model_layers_0/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_19/act/std":0.41516113029298163,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/norm":0.11217617477165713,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/std":0.00041448539535176454,"train/train/tensor_act_model_layers_43_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/mean":9.951181709766388e-07,"train/train/tensor_act_model_layers_17_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/mean":2.731103450059891e-07,"train/train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer__model_layers_67/param/max_abs":1,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_mlp_gate_proj_weight/max_abs":0.00201416015625,"train/train/tensor_act_model_layers_46_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_85/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_85_self_attn_q_proj/std":0.223880861777585,"train/train/tensor_act_model_layers_70/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/max_abs":0.087890625,"train/train/tensor_act_model_layers_86_self_attn_v_proj/mean":0.0077667236328125,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/mean":0.00011968612670898438,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/max_abs":0.08447265625,"train/train/tensor_act_model_layers_18_post_attention_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_11/grad/norm":0.4158142534570563,"train/train/tensor_act_model_layers_63_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_69_self_attn_q_proj/max_abs":1.125,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/std":0.0005340680995047918,"train/train/tensor_act_model_layers_21_self_attn_k_proj/max_abs":1.0625,"train/train/tensor_act_model_layers_90_self_attn_q_proj/norm":1319.2183371760195,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/std":7.28839824937334e-05,"train/train/tensor_act_model_layers_75_self_attn/mean":0.0004374384880065918,"train/train/layer__model_layers_81/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/std":0.00010626033928519196,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_45_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn_q_proj/norm":1280.4476281036243,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn_k_proj/mean":0.0007429122924804688,"train/train/tensor_act_model_layers_72_input_layernorm/max_abs":4.71875,"train/train/layer_model_layers_46/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/mean":-1.2814998626708984e-06,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_13/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/norm":0.02356858155264453,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/mean":-4.601478576660156e-05,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_79/act/norm":9272.805388729943,"train/train/tensor_act_model_rotary_emb/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/mean":3.739260137081146e-07,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/mean":1.862645149230957e-05,"train/train/tensor_param_model_layers_75_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/max_abs":0.00016117095947265625,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/max_abs":0.00010442733764648438,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/mean":0.0001850128173828125,"train/train/tensor_act_model_layers_67_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/norm":0.16530610099190676,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_k_proj/max_abs":1.140625,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/std":0.00024154452785192363,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/mean":3.6954879760742188e-06,"train/train/tensor_act_model_layers_6_mlp_down_proj/std":0.014709983738650977,"train/train/tensor_param_model_layers_21_mlp_gate_proj_weight/mean":7.343292236328125e-05,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/max_abs":0.080078125,"train/train/tensor_act_model_layers_33_self_attn_o_proj/max_abs":0.2177734375,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_down_proj/std":0.015625380809680012,"train/train/tensor_act_model_layers_75_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_mlp/std":0.015869591286343365,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_13_self_attn_o_proj/norm":278.4262211481682,"train/train/layer__model_layers_66/param/std":0.04423943872408442,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/norm":0.06407435702873128,"train/train/tensor_param_model_layers_86_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47/norm":1960.4109104439713,"train/train/tensor_act_model_layers_53_input_layernorm/max_abs":4.84375,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/norm":2.5625,"train/train/layer_model_layers_78/grad/max_abs":0.004150390625,"train/train/tensor_act_model_layers_51_self_attn_o_proj/max_abs":0.2314453125,"train/train/tensor_act_model_layers_32_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_mlp_down_proj/std":0.015335860190988045,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_39_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/max_abs":0.00015163421630859375,"train/train/tensor_act_model_layers_85_post_attention_layernorm/std":1.0000018099073706,"train/train/tensor_act_model_layers_44_mlp_gate_proj/norm":1862.8481575481203,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/std":0.02001953125,"train/train/layer_model_layers_45/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_q_proj/max_abs":1.0546875,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/mean":-3.1068921089172363e-06,"train/train/tensor_act_model_layers_44/std":0.32471198072820134,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/max_abs":0.08251953125,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_k_proj/std":0.22363828252574217,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_53/act/mean":-0.0009548855679375785,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_gate_proj_weight/mean":-1.0263174772262573e-06,"train/train/tensor_act_model_layers_64_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_self_attn_k_proj/max_abs":1.046875,"train/train/layer__model_layers_60/param/mean":0.0015322823606303626,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_up_proj/max_abs":1.2265625,"train/train/tensor_act_model_layers_90_mlp_down_proj/mean":-0.0003209114074707031,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/mean":0.00015735626220703125,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp/std":0.015244750942809374,"train/train/tensor_act_model_layers_81_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_mlp/std":0.015488088323518473,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/max_abs":0.003875732421875,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_9_mlp_gate_proj_weight/max_abs":0.0020294189453125,"train/train/tensor_grad_model_layers_15_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/mean":-3.3248215913772583e-07,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp_up_proj/max_abs":1.0859375,"train/train/tensor_act_model_layers_45_self_attn_o_proj/std":0.05035436493330927,"train/train/tensor_param_model_layers_27_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_0/act/frac_near_user_limit":0,"train/train/layer_model_layers_85/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn_o_proj/max_abs":0.2275390625,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/norm":0.025623355324231507,"train/train/layer__model_layers_37/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/mean":0.000125885009765625,"train/train/layer_model_layers_26/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/mean":4.6253204345703125e-05,"train/train/tensor_act_model_layers_61_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_48_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn/max_abs":0.2236328125,"train/train/tensor_grad_model_layers_79_mlp_gate_proj_weight/std":6.52279041846715e-05,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/max_abs":0.0020294189453125,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/max_abs":0.0013427734375,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_49_mlp_up_proj/norm":1833.0475884345867,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/norm":0.09417279592050772,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_25_mlp/norm":88.08356356843734,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/mean":-2.2258609533309937e-07,"train/train/tensor_act_model_layers_7/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp/std":0.015671192919477207,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_90_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_77/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_18_mlp_gate_proj/std":0.22851952931564978,"train/train/tensor_act_model_layers_48_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/norm":0.0019837306076470862,"train/train/tensor_act_model_layers_47/max_abs":1.703125,"train/train/tensor_act_model_layers_22_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/mean":-2.775341272354126e-06,"train/train/layer__model_layers_71/param/mean":0.0015014077125585024,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/std":0.0006076160758749251,"train/train/tensor_act_model_layers_19_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/mean":-1.7386628314852715e-07,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/std":0.02001953125,"train/train/layer__model_layers_32/param/mean":0.0016430037999859838,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/max_abs":0.000713348388671875,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/max_abs":0.080078125,"train/train/tensor_act_model_layers_58/std":0.37745067135614674,"train/train/tensor_act_model_layers_47_input_layernorm/max_abs":4.78125,"train/train/tensor_act_model_layers_32_input_layernorm/norm":5792.565063477112,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/norm":0.10186926655930058,"train/train/tensor_act_model_layers_5_self_attn_v_proj/mean":0.00039386749267578125,"train/train/tensor_act_model_layers_67_post_attention_layernorm/norm":5792.588623048197,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/mean":5.7697296142578125e-05,"train/train/tensor_act_model_layers_22/norm":1345.6866147035093,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/mean":1.710322976578027e-07,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/max_abs":0.0001468658447265625,"train/train/tensor_act_model_layers_51_self_attn/mean":0.0020809173583984375,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/max_abs":0.002197265625,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/max_abs":0.08203125,"train/train/tensor_param_model_layers_84_mlp_gate_proj_weight/max_abs":0.08056640625,"train/train/layer_model_layers_10/act/max_abs":4.9375,"train/train/tensor_act_model_layers_16_self_attn_k_proj/max_abs":1.1328125,"train/train/tensor_act_model_layers_49_mlp/norm":90.89383213770765,"train/train/layer_model_layers_37/act/std":0.4181063860938998,"train/train/layer_model_layers_64/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/norm":2.5625,"train/train/layer__model_layers_5/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp_down_proj/max_abs":0.0947265625,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/mean":2.530869096517563e-07,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/std":0.020263671875,"train/train/layer__model_layers_10/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/max_abs":0.003753662109375,"train/train/tensor_act_model_layers_27_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/mean":-7.331371307373047e-06,"train/train/tensor_act_model_layers_58_mlp/std":0.01580956405218063,"train/train/tensor_act_model_layers_83_self_attn_k_proj/mean":-0.0121002197265625,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/std":8.179529961757352e-05,"train/train/tensor_act_model_layers_75/mean":0.0011692047119140625,"train/train/tensor_act_model_embed_tokens/mean":-4.112720489501953e-05,"train/train/tensor_act_model_layers_50/mean":-0.0030059814453125,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/norm":0.00011020123512246734,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/std":0.020263671875,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/mean":2.5197550712618977e-09,"train/train/tensor_act_model_layers_4_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/norm":2.53125,"train/train/tensor_act_model_layers_37_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_gate_proj_weight/max_abs":0.08251953125,"train/train/tensor_act_model_layers_16_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_93/param/std":0.0442430971373114,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/max_abs":0.0038604736328125,"train/train/tensor_act_model_layers_44_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/mean":0.00010728836059570312,"train/train/layer_model_layers_33/act/norm":9049.921035553942,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/norm":3.65625,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_35_self_attn_q_proj/norm":1249.4929350435534,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/max_abs":0.004852294921875,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/max_abs":0.08447265625,"train/train/tensor_grad_model_layers_74_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/std":0.0034021991366335754,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_51_self_attn_q_proj/mean":0.0115814208984375,"train/train/tensor_act_model_layers_49_mlp_down_proj/std":0.015671022464751654,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/std":9.467857703346179e-05,"train/train/tensor_act_model_layers_69_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_self_attn_o_proj/norm":298.7526779173787,"train/train/tensor_act_model_layers_47_mlp_down_proj/mean":5.5149197578430176e-05,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/max_abs":0.0947265625,"train/train/layer__model_layers_86/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_gate_proj_weight/max_abs":0.091796875,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/mean":4.410743713378906e-05,"train/train/tensor_act_model_layers_42_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_58/grad/std":0.00020773300848936563,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_59/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn_k_proj/mean":0.0139312744140625,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/norm":0.0020785304925161376,"train/train/tensor_act_model_layers_65_input_layernorm/norm":5792.592529297518,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/max_abs":5.334615707397461e-06,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/std":0.02001953125,"train/train/layer_model_layers_21/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_16/act/mean":-0.010654194014413016,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_mlp_gate_proj_weight/mean":-2.523884177207947e-07,"train/train/tensor_param_model_layers_19_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_89_self_attn_q_proj/norm":1297.616725187384,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_mlp_gate_proj_weight/mean":6.73346221446991e-07,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/max_abs":2.86102294921875e-06,"train/train/tensor_grad_model_layers_23_mlp_gate_proj_weight/mean":7.124617695808411e-07,"train/train/tensor_act_model_layers_17_self_attn_v_proj/mean":-0.0021457672119140625,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/max_abs":0.0849609375,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_43/param/norm":17.934165086330335,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_37_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_89_mlp_gate_proj_weight/max_abs":0.00121307373046875,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/mean":0.0001678466796875,"train/train/tensor_act_model_layers_27_post_attention_layernorm/mean":-0.03338623046875,"train/train/tensor_act_model_layers_75_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/mean":-9.913492249324918e-10,"train/train/tensor_act_model_layers_43_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_92_mlp_up_proj/mean":0.0003936290740966797,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/norm":0.00020110829735944707,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/max_abs":0.00341796875,"train/train/tensor_grad_model_layers_76_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/norm":1337.33448477063,"train/train/tensor_act_model_layers_75_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/mean":-2.0954757928848267e-07,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/mean":7.390975952148438e-05,"train/train/tensor_act_model_layers_32_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/norm":0.11660571322569302,"train/train/layer_model_layers_65/grad/max_abs":0.004302978515625,"train/train/layer__model_layers_21/param/mean":0.0015738214977817863,"train/train/layer__model_layers_70/param/std":0.04423802237143109,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_post_attention_layernorm/mean":-0.05780029296875,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/std":0.0004057077284286331,"train/train/tensor_param_model_layers_47_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_11_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/mean":1.1257827281951904e-05,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/mean":1.0281801223754883e-06,"train/train/tensor_act_model_layers_50_mlp_up_proj/std":0.2312037660157148,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_37_self_attn_o_proj/std":0.04828006581961424,"train/train/layer_model_layers_22/grad/mean":8.099531656234306e-07,"train/train/tensor_act_model_layers_22_input_layernorm/mean":-0.05828857421875,"train/train/layer_model_layers_4/act/max_abs":5.40625,"train/train/tensor_act_model_layers_37/mean":-0.0020656585693359375,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/mean":9.953975677490234e-06,"train/train/tensor_act_model_layers_55_self_attn_q_proj/norm":1315.1130948476696,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp_up_proj/std":0.22412372140341344,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/mean":-1.2500095181167126e-08,"train/train/tensor_act_model_layers_34_input_layernorm/mean":-0.022674560546875,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/std":8.300419321586501e-05,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn_q_proj/std":0.22096583788225918,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_65_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/mean":5.129724740982056e-06,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_9_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/std":0.0201416015625,"train/train/layer_model_layers_46/grad/std":0.00022907747675217078,"train/train/tensor_act_model_layers_28_input_layernorm/mean":-0.034423828125,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/std":2.4717595155092404e-06,"train/train/tensor_act_model_layers_78_mlp_up_proj/mean":-0.00701904296875,"train/train/tensor_act_model_layers_75_mlp_down_proj/norm":91.84183844950404,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/mean":8.949427865445614e-10,"train/train/tensor_act_model_layers_91_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_25_self_attn_o_proj/mean":-0.001499176025390625,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_k_proj/std":0.22950259081108038,"train/train/tensor_act_model_layers_79_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/norm":0.00020230599676512967,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/std":4.865694683230895e-05,"train/train/tensor_grad_model_layers_8_mlp_gate_proj_weight/mean":-7.505295798182487e-07,"train/train/layer__model_layers_1/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_self_attn_v_proj/mean":-0.00063323974609375,"train/train/tensor_act_model_layers_20_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_o_proj/max_abs":0.2275390625,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/mean":-4.267692565917969e-05,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_8_mlp_gate_proj/max_abs":1.3203125,"train/train/tensor_act_model_layers_42_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/std":0.0201416015625,"train/train/layer__model_layers_47/param/max_abs":1,"train/train/tensor_act_model_layers_75_self_attn_k_proj/norm":1362.5445110072833,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp/norm":89.57978686088487,"train/train/tensor_grad_model_layers_0_mlp_gate_proj_weight/std":0.0008585403966009799,"train/train/tensor_act_model_layers_73_self_attn_k_proj/mean":-0.0036773681640625,"train/train/tensor_act_model_layers_12_mlp_down_proj/mean":0.00024651363492012024,"train/train/layer__model_layers_6/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_post_attention_layernorm/max_abs":4.625,"train/train/tensor_act_model_layers_21_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/norm":0.0018717961891691634,"train/train/layer__model_layers_67/param/std":0.044234975128875774,"train/train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_31_mlp/mean":0.0007619857788085938,"train/train/tensor_param_model_layers_37_mlp_gate_proj_weight/mean":-0.0001087188720703125,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/max_abs":0.09228515625,"train/train/layer_model_layers_27/grad/mean":-2.960519728580615e-07,"train/train/layer_model_layers_40/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19/norm":1239.800813554469,"train/train/tensor_act_model_layers_49_self_attn/max_abs":0.2353515625,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/std":2.5227385068394682e-06,"train/train/layer__model_layers_64/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_mlp_gate_proj_weight/mean":-2.608867362141609e-07,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/norm":0.026770648522734832,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/std":0.00048981271407326,"train/train/tensor_act_model_layers_54_self_attn_o_proj/mean":-0.00246429443359375,"train/train/tensor_act_model_layers_76_self_attn_q_proj/std":0.22803470997378947,"train/train/tensor_act_model_layers_12_self_attn_o_proj/std":0.04834108530934276,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/mean":6.239861249923706e-08,"train/train/tensor_act_model_layers_44/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_post_attention_layernorm/std":1.0000003449385393,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/mean":-1.7636921256780624e-08,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/max_abs":6.645917892456055e-06,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/std":0.0006395574913495453,"train/train/tensor_act_model_layers_51_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/max_abs":0.09228515625,"train/train/layer_model_layers_75/grad/max_abs":0.004730224609375,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/max_abs":0.00396728515625,"train/train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_mlp_gate_proj_weight/max_abs":0.0859375,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/std":8.280926233453971e-05,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_v_proj/norm":1337.872989211791,"train/train/layer__model_layers_11/param/norm":17.941807246362476,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_30_input_layernorm/std":1.000000989064085,"train/train/tensor_act_model_layers_89_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_74_post_attention_layernorm/max_abs":4.65625,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/max_abs":0.07861328125,"train/train/tensor_act_model_layers_40_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn_o_proj/norm":292.288934025549,"train/train/layer__model_layers_54/param/norm":17.927500206996932,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/norm":0.0024139891377199614,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/mean":-5.173683166503906e-05,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_9_mlp_down_proj/mean":0.0007171630859375,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/max_abs":4.9173831939697266e-06,"train/train/tensor_param_model_layers_81_mlp_gate_proj_weight/max_abs":0.076171875,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_mlp_gate_proj_weight/norm":0.028699376899321914,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/std":0.00023840174578309862,"train/train/tensor_act_model_layers_45_self_attn/mean":-0.00024832645431160927,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_norm/std":1.0000013913949586,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_o_proj/max_abs":0.234375,"train/train/tensor_param_model_layers_38_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_29_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/std":5.268216164528815e-07,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/max_abs":0.076171875,"train/train/tensor_act_model_layers_53_self_attn_o_proj/norm":268.64735670760024,"train/train/tensor_act_model_layers_91_mlp_gate_proj/mean":-0.00537109375,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/max_abs":0.0059814453125,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_15_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn/std":0.04919603072127065,"train/train/tensor_act_model_layers_31_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_up_proj/mean":0.0011909008026123047,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_86/param/max_abs":1,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/mean":3.5521225072443485e-08,"train/train/layer_model_layers_91/grad/norm":0.1375912671416459,"train/train/tensor_grad_model_layers_6_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80/max_abs":2.203125,"train/train/tensor_act_model_layers_63_mlp_gate_proj/std":0.22827223915342504,"train/train/tensor_param_model_layers_8_mlp_gate_proj_weight/norm":3.59375,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_33_self_attn_q_proj/norm":1300.4107309542524,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/mean":-3.236345946788788e-07,"train/train/tensor_act_model_layers_81_self_attn_v_proj/mean":0.00415802001953125,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/max_abs":0.07470703125,"train/train/tensor_act_model_layers_66/mean":-0.0012874007225036621,"train/train/tensor_act_model_layers_38_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29/max_abs":1.328125,"train/train/tensor_act_model_layers_28_self_attn_v_proj/max_abs":1.140625,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/mean":-0.000217437744140625,"train/train/tensor_act_model_layers_74_mlp_up_proj/norm":1849.107823051581,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/mean":0.0002422332763671875,"train/train/tensor_act_model_layers_38_mlp_up_proj/norm":1839.419492553372,"train/train/tensor_act_model_layers_34_mlp/max_abs":0.0869140625,"train/train/tensor_param_model_layers_20_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_v_proj/norm":1283.769881481798,"train/train/tensor_act_model_layers_57_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/max_abs":5.990266799926758e-06,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/norm":3.59375,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/max_abs":0.0947265625,"train/train/tensor_act_model_layers_82_input_layernorm/max_abs":4.78125,"train/train/tensor_grad_model_layers_89_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/max_abs":0.083984375,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/mean":8.106231689453125e-05,"train/train/tensor_act_model_layers_60/norm":2228.426112522862,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/norm":0.000790896778391565,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/max_abs":0.08251953125,"train/train/layer_model_layers_16/grad/norm":0.3681243210233628,"train/train/layer_model_layers_88/grad/std":0.0001757662461428349,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/max_abs":3.2782554626464844e-06,"train/train/tensor_act_model_layers_81_mlp/norm":89.49962827904578,"train/train/tensor_act_model_layers_73_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_mlp_up_proj/mean":0.0031414031982421875,"train/train/tensor_act_model_layers_37_mlp/mean":0.00010579824447631836,"train/train/tensor_act_model_layers_40_mlp_up_proj/std":0.22339186909575065,"train/train/tensor_act_model_layers_43_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/std":1.4836922241410105e-06,"train/train/tensor_act_model_layers_83_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43/max_abs":1.640625,"train/train/tensor_act_model_layers_35_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/std":0.00031648704344573096,"train/train/tensor_act_model_layers_57_post_attention_layernorm/max_abs":4.625,"train/train/tensor_act_model_layers_50_mlp_up_proj/norm":1897.8305294142338,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_65/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/max_abs":0.00013256072998046875,"train/train/tensor_act_model_layers_71/norm":2439.767842366291,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/mean":3.3080577850341797e-06,"train/train/tensor_grad_model_layers_26_mlp_gate_proj_weight/max_abs":0.001251220703125,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_43_mlp_down_proj/std":0.015580080258191584,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_91_post_attention_layernorm/std":1.0000036278440934,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/mean":-8.73953104019165e-06,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/max_abs":0.08349609375,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/norm":0.29637022773829436,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/mean":-5.681067705154419e-08,"train/train/layer_model_layers_58/grad/mean":2.36559657063248e-07,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_75_mlp_gate_proj/norm":1878.1603036746492,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/max_abs":0.0791015625,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/max_abs":0.0002574920654296875,"train/train/tensor_act_model_layers_46_mlp_up_proj/norm":1835.7399494726894,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/norm":2.53125,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/mean":2.4065375328063965e-06,"train/train/tensor_act_model_layers_89_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/max_abs":0.08154296875,"train/train/tensor_act_model_layers_78_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/max_abs":8.165836334228516e-06,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_77_input_layernorm/mean":0.0040435791015625,"train/train/tensor_act_model_layers_37_self_attn_k_proj/max_abs":1.0234375,"train/train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_8/act/std":0.41250407856073296,"train/train/layer_model_layers_8/grad/mean":-2.0819799017448807e-06,"train/train/tensor_act_model_layers_26_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/max_abs":0.000881195068359375,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/std":0.0004391278969847076,"train/train/tensor_param_model_layers_73_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_36/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_up_proj/mean":-0.004405975341796875,"train/train/layer_model_layers_62/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/norm":0.024908775546157058,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_input_layernorm/max_abs":4.71875,"train/train/tensor_grad_model_layers_75_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_post_attention_layernorm/norm":5792.5510253924,"train/train/tensor_act_model_layers_58_post_attention_layernorm/max_abs":4.625,"train/train/tensor_param_model_layers_91_mlp_gate_proj_weight/mean":-3.123283386230469e-05,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_72_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_1_mlp_down_proj/std":0.01556400094021893,"train/train/tensor_grad_model_layers_41_mlp_gate_proj_weight/norm":0.03725285149227398,"train/train/tensor_act_model_layers_3_self_attn_o_proj/norm":226.96487138190938,"train/train/tensor_act_model_layers_25_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_input_layernorm/mean":0.002998828887939453,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/std":9.459014262091551e-05,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/max_abs":0.0771484375,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/max_abs":0.08837890625,"train/train/tensor_act_model_layers_58/frac_near_user_limit":0,"train/train/layer_model_layers_0/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_mlp_gate_proj_weight/mean":4.563480615615845e-08,"train/train/tensor_param_model_layers_41_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_mlp_gate_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/norm":0.03255983896053989,"train/train/tensor_act_model_layers_8_mlp_up_proj/max_abs":1.25,"train/train/tensor_param_model_layers_48_input_layernorm_weight/max_abs":1,"train/train/layer__model_layers_61/param/max_abs":1,"train/train/layer__model_layers_50/param/max_abs":1,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn/norm":290.40290013720914,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_q_proj/mean":0.004057884216308594,"train/train/tensor_act_model_layers_89_mlp/norm":90.57736296760045,"train/train/layer_model_layers_69/grad/std":0.0002048875048957465,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/mean":-2.316199243068695e-06,"train/train/layer_model_layers_66/act/std":0.42538987454437915,"train/train/tensor_act_model_layers_61_input_layernorm/norm":5792.589477540826,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/norm":0.12180757858488983,"train/train/tensor_act_model_layers_10_self_attn/mean":-0.001552581787109375,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/max_abs":6.288290023803711e-06,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/std":3.56328666419123e-05,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/mean":0.000133514404296875,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_45_self_attn_o_proj/max_abs":0.236328125,"train/train/tensor_act_model_layers_69_mlp_up_proj/max_abs":1.125,"train/train/tensor_act_model_layers_34_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_v_proj/std":0.218753088243858,"train/train/tensor_act_model_layers_46/max_abs":1.6953125,"train/train/tensor_act_model_layers_52_self_attn_q_proj/mean":-0.00791168212890625,"train/train/tensor_act_model_layers_44_mlp/std":0.016083084552597656,"train/train/tensor_act_model_layers_0_self_attn/norm":50.61179826723799,"train/train/tensor_act_model_layers_36_mlp_up_proj/mean":-0.0118865966796875,"train/train/tensor_act_model_layers_48_mlp/mean":0.00019985437393188477,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/max_abs":6.139278411865234e-06,"train/train/tensor_act_model_layers_41_input_layernorm/max_abs":4.875,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/mean":-1.7113052308559418e-08,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/mean":-2.2411346435546875e-05,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_9_post_attention_layernorm/norm":5792.464233402331,"train/train/tensor_param_model_layers_34_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/std":0.00011569665812688967,"train/train/tensor_act_model_layers_12_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/mean":1.3872981071472168e-05,"train/train/layer_model_layers_15/grad/norm":0.3569110795244852,"train/train/tensor_act_model_layers_16_mlp_gate_proj/norm":1881.88391065729,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_post_attention_layernorm/mean":-0.016265869140625,"train/train/tensor_act_model_layers_7_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/max_abs":0.0030517578125,"train/train/tensor_act_model_layers_5_self_attn_k_proj/mean":-0.0074005126953125,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/std":6.776032850714144e-07,"train/train/layer__model_layers_30/param/max_abs":1,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/std":0.0004295009344906817,"train/train/tensor_act_model_layers_2_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/mean":-0.0002880096435546875,"train/train/tensor_act_model_layers_63_post_attention_layernorm/norm":5792.59191894864,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/norm":0.12171848713328584,"train/train/tensor_act_model_layers_38_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_25/grad/std":0.0003162523052916387,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/std":2.892483358882093e-05,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_gate_proj_weight/mean":-2.086162567138672e-07,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/mean":0.000118255615234375,"train/train/tensor_param_model_layers_78_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/mean":-4.596586222760379e-09,"train/train/tensor_grad_model_layers_36_mlp_gate_proj_weight/norm":0.039591826815968095,"train/train/tensor_act_model_layers_65_input_layernorm/mean":0.00035858154296875,"train/train/tensor_act_model_layers_75_self_attn/std":0.04864668740597001,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/norm":2.53125,"train/train/tensor_act_model_layers_45_post_attention_layernorm/max_abs":4.84375,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/norm":0.00011773557488346528,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/std":0.00018790963710776573,"train/train/tensor_act_model_layers_12_self_attn_q_proj/max_abs":1.0859375,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/mean":4.4819898903369904e-07,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/norm":0.24586645862228518,"train/train/tensor_act_model_layers_79_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/mean":3.0454248189926147e-06,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/mean":-1.7008278518915176e-07,"train/train/layer_model_layers_3/grad/norm":0.9001279600808811,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/mean":6.51925802230835e-07,"train/train/tensor_act_model_layers_85_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn/std":0.05023454374586558,"train/train/tensor_act_model_layers_13_self_attn/norm":278.4262211481682,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/max_abs":5.507469177246094e-05,"train/train/layer_model_layers_75/act/std":0.4276165151065958,"train/train/tensor_act_model_layers_34_input_layernorm/norm":5792.580078128879,"train/train/tensor_act_model_layers_59_self_attn_q_proj/mean":0.0013904571533203125,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/max_abs":5.543231964111328e-06,"train/train/layer__model_layers_21/param/max_abs":1,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/max_abs":0.00970458984375,"train/train/tensor_act_model_layers_22_input_layernorm/norm":5792.552001956711,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn/max_abs":0.234375,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/norm":0.3319263965873531,"train/train/tensor_act_model_layers_31_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/mean":3.194145392626524e-09,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/mean":0.00017833709716796875,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_85/grad/frac_near_user_limit":0,"train/train/layer__model_layers_82/param/norm":17.933239367686753,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/norm":0.0003209974816847639,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp_gate_proj/max_abs":1.125,"train/train/tensor_act_model_layers_9_self_attn/max_abs":0.2197265625,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_84_mlp_gate_proj_weight/std":6.664688467519234e-05,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/norm":0.0024656168221083036,"train/train/layer_model_layers_62/grad/max_abs":0.005096435546875,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/max_abs":0.0002193450927734375,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/max_abs":0.0028839111328125,"train/train/tensor_param_model_layers_9_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/mean":4.696846008300781e-05,"train/train/tensor_param_model_layers_50_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_18_self_attn_k_proj/std":0.23267115004254038,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_mlp_up_proj/std":0.22681136839818367,"train/train/layer__model_layers_49/param/std":0.04424900127125365,"train/train/tensor_act_model_layers_74_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_18_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/max_abs":0.09228515625,"train/train/tensor_act_model_layers_28_input_layernorm/norm":5792.561645512601,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/max_abs":0.08203125,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/max_abs":0.0859375,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/norm":0.0256232099373054,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/norm":0.028603225868036406,"train/train/tensor_param_model_layers_2_mlp_gate_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_54_mlp_down_proj/norm":90.84606674281565,"train/train/tensor_act_model_layers_15_mlp_gate_proj/norm":1896.9482743422207,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/mean":-0.00013828277587890625,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/mean":6.341934204101562e-05,"train/train/tensor_act_model_layers_55_self_attn_o_proj/max_abs":0.2431640625,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/std":0.0004265936888091744,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/mean":-8.344650268554688e-07,"train/train/tensor_act_model_layers_76_input_layernorm/std":1.0000010226407767,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/norm":0.10525207869046839,"train/train/layer_model_layers_16/act/norm":8987.917375605532,"train/train/tensor_param_model_layers_17_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/norm":0.0016614397080542422,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/std":0.00012808242391607549,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_o_proj/std":0.04388577182808841,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/max_abs":0.00628662109375,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/mean":1.229636836796999e-09,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_post_attention_layernorm/std":1.000003409469727,"train/train/tensor_grad_model_layers_75_mlp_gate_proj_weight/mean":-3.976747393608093e-07,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/mean":-6.246566772460938e-05,"train/train/tensor_act_model_layers_79_self_attn_v_proj/std":0.23120623056719408,"train/train/tensor_param_model_layers_23_mlp_gate_proj_weight/std":0.019775390625,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/norm":0.00012474829244667491,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/norm":0.11351331420982813,"train/train/tensor_act_model_layers_29_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_k_proj/norm":1305.0404014829992,"train/train/tensor_act_model_layers_25_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/mean":4.5122578740119934e-07,"train/train/tensor_act_model_layers_44_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/norm":2.5625,"train/train/layer__model_layers_0/param/max_abs":1,"train/train/tensor_act_model_layers_24_self_attn_q_proj/mean":0.000644683837890625,"train/train/tensor_act_model_layers_63_input_layernorm/std":1.0000089241381986,"train/train/tensor_act_model_layers_61_input_layernorm/mean":-0.01180267333984375,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_70_self_attn_v_proj/max_abs":1.109375,"train/train/tensor_act_model_layers_50_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/max_abs":0.08056640625,"train/train/tensor_act_model_layers_51_input_layernorm/std":1.0000051722029577,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/norm":0.04519866588024613,"train/train/tensor_param_model_layers_74_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn_v_proj/norm":1289.1719113115837,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31/norm":1593.2040392985027,"train/train/tensor_act_model_layers_24_mlp_up_proj/max_abs":1.1640625,"train/train/tensor_act_model_layers_56_self_attn_k_proj/max_abs":1.15625,"train/train/tensor_act_model_layers_12_self_attn_q_proj/std":0.22388018748236072,"train/train/tensor_act_model_layers_76_self_attn_o_proj/std":0.048590320733836546,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn_k_proj/norm":1313.7309217579486,"train/train/tensor_param_model_layers_91_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_78_self_attn_k_proj/std":0.2224227240726549,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/mean":6.103515625e-05,"train/train/tensor_param_model_layers_6_mlp_gate_proj_weight/mean":5.507469177246094e-05,"train/train/tensor_act_model_layers_84_self_attn_k_proj/std":0.2351159415852697,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/max_abs":1.043081283569336e-05,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/std":1.5574058334019703e-06,"train/train/layer_model_layers_88/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_input_layernorm/std":1.0000038947834222,"train/train/tensor_act_model_layers_27_self_attn_k_proj/mean":0.0030117034912109375,"train/train/tensor_param_model_layers_8_mlp_gate_proj_weight/mean":-6.67572021484375e-05,"train/train/tensor_act_model_layers_92/mean":0.00766754150390625,"train/train/tensor_act_model_layers_29_mlp_up_proj/max_abs":1.2265625,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/norm":0.02811495548598897,"train/train/tensor_act_model_layers_36_self_attn/max_abs":0.23046875,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_mlp_gate_proj_weight/std":8.108839682938223e-05,"train/train/tensor_act_model_layers_66_mlp_up_proj/mean":-0.0088958740234375,"train/train/tensor_act_model_layers_30_mlp_gate_proj/mean":-0.005584716796875,"train/train/layer_model_layers_60/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_gate_proj/std":0.22387884140792905,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/norm":2.53125,"train/train/tensor_act_model_layers_83_mlp_up_proj/norm":1874.2331334106,"train/train/tensor_act_model_layers_70_self_attn_o_proj/norm":283.6257155733301,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/max_abs":0.004302978515625,"train/train/tensor_act_model_layers_32_input_layernorm/mean":-0.0306396484375,"train/train/layer_model_layers_69/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_post_attention_layernorm/max_abs":4.75,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7/mean":-0.01007080078125,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/mean":-4.553794860839844e-05,"train/train/tensor_act_model_layers_12_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_v_proj/std":0.22363286702913346,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/std":2.9215474053801185e-05,"train/train/tensor_act_model_layers_54_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15/norm":1079.0220762146325,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_v_proj/norm":1282.477876581672,"train/train/tensor_act_model_layers_5_post_attention_layernorm/std":0.9970830660610555,"train/train/tensor_grad_model_layers_67_mlp_gate_proj_weight/norm":0.028485376174080708,"train/train/tensor_act_model_layers_5_mlp/max_abs":0.0869140625,"train/train/tensor_act_model_layers_60_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_92_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn/frac_near_dtype_limit":0,"train/train/layer__model_layers_9/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/norm":0.024853377725934357,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/mean":3.973022103309631e-06,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/mean":1.1599622666835785e-06,"train/train/tensor_act_model_layers_57_post_attention_layernorm/norm":5792.58349609481,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/std":0.000332619423305896,"train/train/tensor_act_model_layers_50_self_attn_k_proj/max_abs":1.0390625,"train/train/layer_model_layers_13/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_3/act/mean":-0.006652729851858956,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/mean":1.587904989719391e-07,"train/train/tensor_grad_model_layers_55_mlp_gate_proj_weight/max_abs":0.0016021728515625,"train/train/tensor_act_model_layers_65_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_self_attn_q_proj/max_abs":1.1640625,"train/train/tensor_act_model_layers_73_post_attention_layernorm/mean":8.58306884765625e-05,"train/train/layer__model_layers_78/param/std":0.04424679418100066,"train/train/layer__model_layers_80/param/mean":0.0015426849984155617,"train/train/tensor_act_model_layers_30_mlp/max_abs":0.08251953125,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/std":3.543855149173192e-05,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/norm":0.09994450459794481,"train/train/tensor_act_model_layers_28_mlp/norm":89.53416260285036,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/mean":-4.0140002965927124e-07,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/mean":9.119510650634766e-06,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_82_self_attn_v_proj/norm":1345.3146350180803,"train/train/tensor_act_model_layers_81_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/std":0.015686253256347822,"train/train/tensor_act_model_layers_55_self_attn_k_proj/norm":1318.7456802365805,"train/train/tensor_param_model_layers_57_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_68/mean":0.0017881393432617188,"train/train/tensor_act_model_layers_6_input_layernorm/max_abs":5.03125,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_45_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/mean":0.000156402587890625,"train/train/tensor_act_model_layers_7_mlp_down_proj/max_abs":0.091796875,"train/train/tensor_act_model_layers_43_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp/max_abs":0.0791015625,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/std":0.00012849455147682652,"train/train/tensor_act_model_layers_62_self_attn_k_proj/std":0.22754843162005858,"train/train/tensor_grad_model_layers_4_mlp_gate_proj_weight/max_abs":0.0035858154296875,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_64_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_o_proj/mean":0.002838134765625,"train/train/tensor_act_model_layers_81_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_41/act/std":0.4194750272808865,"train/train/tensor_act_model_layers_29_self_attn_o_proj/max_abs":0.2216796875,"train/train/tensor_act_model_layers_85_mlp_down_proj/mean":-0.0006122589111328125,"train/train/tensor_act_model_layers_52/std":0.35694534106853776,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/mean":3.838539123535156e-05,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/max_abs":0.00762939453125,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp/std":0.014969023365321023,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/max_abs":0.005523681640625,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/norm":0.3670809976862119,"train/train/layer_model_layers_69/act/std":0.4253829059804503,"train/train/tensor_act_model_layers_80_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_v_proj/std":0.2309604014396249,"train/train/tensor_act_model_layers_21/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_v_proj/max_abs":1.0546875,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/max_abs":0.0849609375,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/max_abs":0.005462646484375,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/norm":3.609375,"train/train/layer_model_layers_3/grad/max_abs":0.0260009765625,"train/train/layer__model_layers_62/param/std":0.04425681141039765,"train/train/tensor_act_model_layers_39/std":0.31006367671734997,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65/frac_near_dtype_limit":0,"train/train/layer__model_layers_52/param/max_abs":1,"train/train/tensor_param_model_layers_6_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_7_mlp_gate_proj/norm":1859.636729624671,"train/train/tensor_param_model_layers_75_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_9/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_q_proj/max_abs":1.2890625,"train/train/tensor_act_model_layers_92_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/std":0.00012995042741991152,"train/train/tensor_act_model_layers_75_self_attn_q_proj/norm":1314.7239134508127,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn/norm":278.5058196344276,"train/train/layer_model_layers_42/act/std":0.41963250998796525,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/mean":9.250640869140625e-05,"train/train/tensor_act_model_layers_77_mlp_down_proj/std":0.015961466057052225,"train/train/tensor_param_model_layers_67_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_39_mlp_gate_proj/max_abs":1.1953125,"train/train/tensor_act_model_layers_6_mlp/std":0.014709983738650977,"train/train/layer_model_layers_51/grad/max_abs":0.004150390625,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_o_proj/max_abs":0.244140625,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/max_abs":2.6464462280273438e-05,"train/train/tensor_act_model_layers_49_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/std":0.00025607419606309226,"train/train/layer_model_layers_8/grad/std":0.0006148436187277904,"train/train/tensor_act_model_layers_29_mlp/std":0.015367045917187053,"train/train/tensor_act_model_layers_16_mlp_down_proj/norm":90.95531207695426,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_80_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/max_abs":0.08447265625,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_o_proj/max_abs":0.25390625,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/std":0.00014315938440405157,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/mean":3.719329833984375e-05,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/mean":2.0384788513183594e-05,"train/train/tensor_act_model_layers_84_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_68/grad/norm":0.159505079575421,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/std":7.855813660935158e-07,"train/train/tensor_act_model_layers_10_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/max_abs":0.08251953125,"train/train/layer_model_layers_38/grad/std":0.00025912312259706124,"train/train/tensor_grad_model_layers_35_mlp_gate_proj_weight/max_abs":0.0023345947265625,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_65_mlp_down_proj/max_abs":0.08251953125,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_8/std":0.13013015294927088,"train/train/tensor_act_model_layers_38/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_input_layernorm/norm":5792.591674806728,"train/train/tensor_act_model_layers_79_post_attention_layernorm/max_abs":4.6875,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn/norm":286.85756736947855,"train/train/tensor_act_model_layers_18_mlp_down_proj/max_abs":0.08447265625,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/max_abs":0.1005859375,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_k_proj/norm":1341.029094731544,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/max_abs":0.08056640625,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/max_abs":0.07861328125,"train/train/tensor_act_model_layers_2_self_attn_k_proj/max_abs":1.21875,"train/train/tensor_param_model_layers_59_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_54_self_attn_o_proj/std":0.04938035833751627,"train/train/tensor_act_model_layers_69_self_attn_o_proj/mean":-0.0021514892578125,"train/train/tensor_act_model_layers_9_post_attention_layernorm/std":1.0000009443606448,"train/train/tensor_act_model_layers_70_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp_gate_proj/norm":1874.63074707515,"train/train/tensor_act_model_layers_15_mlp_down_proj/norm":94.01638228446947,"train/train/tensor_grad_model_layers_50_mlp_gate_proj_weight/std":9.668779601626909e-05,"train/train/tensor_param_model_layers_13_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_6_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_gate_proj/mean":-0.0013799667358398438,"train/train/tensor_act_model_layers_37_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp_down_proj/mean":-5.638599395751953e-05,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/max_abs":0.07421875,"train/train/tensor_act_model_layers_48_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_q_proj/std":0.21948786124526773,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/max_abs":0.0026702880859375,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/max_abs":0.00054931640625,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/std":0.00038013952919821405,"train/train/tensor_grad_model_layers_74_mlp_gate_proj_weight/max_abs":0.00121307373046875,"train/train/tensor_param_model_layers_75_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_12_mlp_gate_proj_weight/std":0.00020548825672579926,"train/train/tensor_act_model_layers_30_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_v_proj/mean":-0.0115509033203125,"train/train/tensor_act_model_layers_55/mean":-0.0024518966674804688,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_23_mlp_gate_proj_weight/norm":0.05366014332940192,"train/train/tensor_act_model_layers_27_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_mlp_gate_proj_weight/norm":0.04755762870796425,"train/train/tensor_act_model_layers_47_mlp_gate_proj/max_abs":1.0625,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_63_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/max_abs":0.0888671875,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_input_layernorm/norm":5792.596435550636,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/max_abs":0.0849609375,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/mean":-3.484543412923813e-06,"train/train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/max_abs":0.07958984375,"train/train/tensor_act_model_layers_11_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_q_proj/mean":-0.0109405517578125,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/std":0.00036786268435252473,"train/train/tensor_param_model_layers_28_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_o_proj/max_abs":0.2197265625,"train/train/tensor_act_model_layers_3_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_51_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/std":0.0005482095529364584,"train/train/tensor_act_model_layers_92_self_attn_k_proj/max_abs":1.109375,"train/train/tensor_param_model_layers_15_mlp_gate_proj_weight/mean":0.0002155303955078125,"train/train/tensor_act_model_layers_11_mlp/mean":0.00015461444854736328,"train/train/tensor_act_model_layers_68_mlp_up_proj/norm":1840.0585044549646,"train/train/tensor_act_model_layers_55_self_attn/std":0.04913560850043175,"train/train/tensor_param_model_layers_92_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp_down_proj/max_abs":0.0791015625,"train/train/layer__model_layers_8/param/std":0.04425062286111048,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/max_abs":0.0001373291015625,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/max_abs":0.0771484375,"train/train/tensor_act_model_layers_43_self_attn/mean":-0.0018444061279296875,"train/train/tensor_act_model_layers_73_mlp_down_proj/std":0.015228798331347491,"train/train/layer_model_layers_38/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/mean":-3.314018249511719e-05,"train/train/tensor_param_model_layers_92_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn_q_proj/std":0.22583421252306607,"train/train/tensor_act_model_layers_57_self_attn_o_proj/mean":0.00122833251953125,"train/train/tensor_act_model_layers_19_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn_v_proj/std":0.22973780373110558,"train/train/layer_model_layers_12/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/std":3.916602677364853e-05,"train/train/tensor_act_model_layers_3_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_24_mlp_up_proj/mean":-0.00038355588912963867,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/std":1.1564505852771251e-05,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_89/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/mean":-2.372264862060547e-05,"train/train/layer__model_layers_40/param/max_abs":1,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/norm":0.0397673903637372,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/max_abs":0.0771484375,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/norm":0.0308966930104318,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/mean":3.6561687011271715e-09,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/std":0.000455491394110536,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/norm":0.0001291718329398894,"train/train/tensor_act_model_layers_36_mlp_down_proj/std":0.01617447514501086,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/max_abs":1.1324882507324219e-05,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/mean":-0.00023746490478515625,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_55_post_attention_layernorm/max_abs":4.875,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38/norm":1770.6915732380617,"train/train/tensor_param_model_layers_13_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp_up_proj/max_abs":1.1640625,"train/train/tensor_act_model_layers_82_mlp/mean":6.339699029922485e-05,"train/train/tensor_act_model_layers_63_self_attn/std":0.048890123159887076,"train/train/tensor_param_model_layers_60_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/norm":0.026316983500170536,"train/train/tensor_act_model_layers_91/std":0.48340517121258497,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_mlp_gate_proj_weight/mean":-5.7697296142578125e-05,"train/train/layer__model_layers_51/param/mean":0.001505073630679602,"train/train/tensor_param_model_layers_78_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/mean":3.039836883544922e-06,"train/train/tensor_act_model_layers_6_mlp_down_proj/norm":85.25050333843588,"train/train/tensor_act_model_layers_87_mlp/mean":-0.0002739429473876953,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/mean":-1.1518597602844238e-05,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/max_abs":0.00168609619140625,"train/train/tensor_act_model_layers_72/norm":2467.5112733981596,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/mean":-2.5238841772079468e-06,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_22/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp_up_proj/norm":1887.5914266675395,"train/train/tensor_act_model_layers_0_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp/norm":90.49846609600961,"train/train/tensor_act_model_layers_39_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36/norm":1719.4382256361748,"train/train/tensor_act_model_layers_63_mlp_down_proj/mean":0.00018638372421264648,"train/train/tensor_act_model_layers_71_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn_o_proj/norm":270.34992872621643,"train/train/layer_model_layers_46/act/mean":-0.001109631998198373,"train/train/layer_model_layers_89/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_80_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/max_abs":0.003936767578125,"train/train/tensor_act_model_layers_84/norm":2675.6491540674197,"train/train/tensor_act_model_layers_26/mean":-0.0110626220703125,"train/train/tensor_grad_model_layers_45_mlp_gate_proj_weight/norm":0.034742463358047045,"train/train/tensor_act_model_layers_47_mlp_down_proj/max_abs":0.08935546875,"train/train/tensor_act_model_layers_66_self_attn_o_proj/norm":283.6611416763035,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/std":2.9677336360745874e-05,"train/train/tensor_act_model_layers_15_mlp_gate_proj/std":0.23169059522003607,"train/train/tensor_act_model_layers_20/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/max_abs":0.08935546875,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/std":6.964847760134285e-05,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/mean":-3.136228770017624e-07,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/max_abs":0.08984375,"train/train/layer__model_layers_67/param/norm":17.921541245802494,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/norm":0.12325716970454677,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/max_abs":0.00012302398681640625,"train/train/layer_model_layers_82/grad/std":0.0001822611777586348,"train/train/tensor_act_model_layers_91_mlp_gate_proj/max_abs":1.1875,"train/train/layer_model_layers_92/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_83/act/norm":9300.583699826459,"train/train/tensor_act_model_layers_93_post_attention_layernorm/norm":5792.604858404708,"train/train/tensor_param_model_layers_88_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_14_post_attention_layernorm/mean":-0.05462646484375,"train/train/layer__model_layers_16/param/max_abs":1,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/std":8.239260733364993e-05,"train/train/layer_model_layers_1/act/max_abs":5.3125,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/norm":2.53125,"train/train/tensor_act_model_layers_23_mlp/mean":-0.0003509521484375,"train/train/tensor_act_model_layers_22_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_gate_proj/mean":-0.004791259765625,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/max_abs":3.4570693969726562e-06,"train/train/layer_model_layers_63/grad/std":0.00020285293853703254,"train/train/tensor_param_model_layers_25_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_84_self_attn/std":0.04962322336792662,"train/train/layer_model_layers_41/act/max_abs":4.96875,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/mean":1.481175422668457e-05,"train/train/tensor_act_model_layers_27_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp_down_proj/max_abs":0.080078125,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_self_attn_o_proj/norm":285.6391186456364,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/std":6.277027957368208e-07,"train/train/tensor_act_model_layers_60_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp_gate_proj/max_abs":1.234375,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/mean":9.965896606445312e-05,"train/train/tensor_act_model_layers_78_mlp_gate_proj/std":0.22754278568134942,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn/std":0.048037360582338394,"train/train/tensor_act_model_layers_17_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_mlp_gate_proj_weight/std":9.348555664868975e-05,"train/train/tensor_act_model_layers_2_mlp_gate_proj/max_abs":1.2265625,"train/train/tensor_act_model_layers_28_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_60/grad/norm":0.16032134046449537,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/max_abs":0.083984375,"train/train/tensor_act_model_layers_53_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_37_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn_k_proj/mean":0.0041637420654296875,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/norm":0.03577555781901888,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_40_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_54_mlp_up_proj/max_abs":1.109375,"train/train/tensor_act_model_layers_83/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/std":7.75076804103133e-05,"train/train/tensor_act_model_layers_92_mlp/max_abs":0.0869140625,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/std":7.11834341148408e-05,"train/train/tensor_act_model_layers_55_self_attn_k_proj/max_abs":1.2734375,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_72_self_attn_q_proj/max_abs":1.0078125,"train/train/tensor_act_model_layers_52/norm":2067.7874047745954,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/mean":4.651956260204315e-07,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_q_proj/mean":-0.002307891845703125,"train/train/tensor_param_model_layers_31_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/mean":-0.0002460479736328125,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/mean":-7.867813110351562e-05,"train/train/tensor_act_model_layers_6_self_attn_o_proj/mean":-0.003711700439453125,"train/train/tensor_act_model_layers_48/std":0.34033716648919293,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/std":0.00010928554063299063,"train/train/tensor_act_model_layers_50_self_attn_o_proj/norm":277.55999526932055,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/mean":-0.000232696533203125,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/mean":4.7404319047927856e-07,"train/train/layer_model_layers_90/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_mlp/norm":89.25947714919363,"train/train/layer_model_layers_81/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp/max_abs":0.08154296875,"train/train/tensor_act_model_layers_26_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_76_self_attn/norm":281.4429203156817,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/max_abs":0.004730224609375,"train/train/tensor_act_model_layers_60_post_attention_layernorm/mean":-0.0118408203125,"train/train/tensor_act_model_layers_21_self_attn/mean":0.00026726722717285156,"train/train/tensor_act_model_layers_35_mlp/max_abs":0.0830078125,"train/train/tensor_act_model_layers_50_post_attention_layernorm/mean":-0.007740020751953125,"train/train/tensor_act_model_layers_16_self_attn_v_proj/std":0.23291602726475377,"train/train/tensor_act_model_layers_90_mlp_gate_proj/mean":0.001046895980834961,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/mean":-2.130400389432907e-08,"train/train/tensor_act_model_layers_69_post_attention_layernorm/max_abs":4.625,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/max_abs":0.0003223419189453125,"train/train/tensor_act_model_layers_66_input_layernorm/mean":0.0015716552734375,"train/train/tensor_act_model_layers_27_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_input_layernorm/mean":-0.04876708984375,"train/train/tensor_act_model_layers_63_mlp_up_proj/norm":1855.3972756420155,"train/train/tensor_act_model_layers_11_self_attn_q_proj/std":0.2246138188710332,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/mean":0.0002574920654296875,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_mlp_up_proj/max_abs":1.1015625,"train/train/tensor_act_model_layers_67_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_mlp_down_proj/norm":88.10078045245818,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/std":0.0023991753736620153,"train/train/layer_model_layers_66/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/max_abs":0.0859375,"train/train/tensor_act_model_layers_72_mlp_down_proj/norm":89.93018348327459,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_56/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_62/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_input_layernorm/norm":5792.597167973186,"train/train/layer_model_layers_9/act/norm":8930.10597803691,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_11/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_input_layernorm/max_abs":4.59375,"train/train/tensor_act_model_layers_12_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/max_abs":0.00010395050048828125,"train/train/tensor_act_model_layers_26/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/mean":2.8014183044433594e-05,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/mean":-6.288290023803711e-06,"train/train/tensor_grad_model_layers_27_mlp_gate_proj_weight/max_abs":0.0024871826171875,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/mean":1.1371448636054993e-06,"train/train/tensor_grad_model_layers_34_mlp_gate_proj_weight/max_abs":0.002288818359375,"train/train/tensor_param_model_layers_75_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_53_mlp/norm":88.05043513674913,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/norm":0.14137407642875685,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/max_abs":0.076171875,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_88/act/mean":0.0012291116373879568,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/mean":3.121793270111084e-06,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/norm":0.08592119929298914,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/norm":0.0023636474995513855,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/mean":0.000335693359375,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_53_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_91_mlp/max_abs":0.08447265625,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85/max_abs":2.1875,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/max_abs":0.087890625,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/mean":7.195631042122841e-07,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/norm":0.14034157908324987,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/mean":0.0001659393310546875,"train/train/tensor_act_model_layers_70_post_attention_layernorm/norm":5792.593994142379,"train/train/tensor_act_model_layers_31_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/max_abs":0.0263671875,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/mean":2.419576048851013e-06,"train/train/tensor_act_model_layers_85_mlp_down_proj/max_abs":0.08447265625,"train/train/tensor_param_model_layers_40_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_29/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_q_proj/mean":-0.0121307373046875,"train/train/layer_model_layers_8/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/norm":0.13562533611484487,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/norm":0.0021053846075711675,"train/train/tensor_act_model_layers_48_mlp_gate_proj/norm":1849.7428966889543,"train/train/tensor_act_model_layers_64_self_attn/std":0.05054102582989184,"train/train/tensor_act_model_layers_35_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/norm":0.0001089231136985632,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/std":0.00019840841658385274,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/std":1.4022509796448301e-06,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/max_abs":1.3232231140136719e-05,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/mean":4.616595106199384e-09,"train/train/tensor_act_model_layers_13_self_attn_k_proj/std":0.2260767069405714,"train/train/layer_model_layers_17/grad/frac_near_dtype_limit":0,"train/train/layer__model_layers_1/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_down_proj/norm":91.11790881556891,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/max_abs":0.004150390625,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/mean":-4.2980536818504333e-07,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/max_abs":0.00107574462890625,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/mean":-1.154148776549846e-09,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/norm":0.1162152970722447,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/norm":0.00023590286520532543,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/norm":2.546875,"train/train/layer_model_layers_1/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_25_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_54/grad/mean":7.83078740794043e-08,"train/train/layer_model_layers_58/act/norm":9169.759947179018,"train/train/tensor_param_model_layers_70_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/max_abs":0.0859375,"train/train/tensor_param_model_layers_19_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_23_post_attention_layernorm/max_abs":4.65625,"train/train/tensor_act_model_layers_88_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_87_post_attention_layernorm/norm":5792.598388677859,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn_v_proj/mean":9.465217590332031e-05,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_34/act/norm":9058.82473625456,"train/train/layer_model_layers_50/grad/max_abs":0.00482177734375,"train/train/tensor_grad_model_layers_36_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/norm":0.01462065923902773,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/mean":3.695487976074219e-05,"train/train/layer__model_layers_37/param/mean":0.0016294768001657567,"train/train/layer__model_layers_42/param/std":0.044226554566037225,"train/train/tensor_act_model_layers_91_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/mean":1.5676021575927734e-05,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/norm":0.0017214304845895373,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/std":2.1908215785261696e-06,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/max_abs":3.0249357223510742e-06,"train/train/tensor_act_model_layers_86_self_attn_k_proj/mean":0.0018150806427001953,"train/train/tensor_param_model_layers_39_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/norm":0.26703627919237344,"train/train/tensor_param_model_layers_76_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn/std":0.05139277155579566,"train/train/tensor_act_model_layers_93_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp/max_abs":0.0869140625,"train/train/tensor_act_model_layers_80_mlp_up_proj/std":0.22656462676698044,"train/train/tensor_act_model_layers_22_self_attn_v_proj/max_abs":1.171875,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/mean":0.0004367828369140625,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_self_attn_o_proj/max_abs":0.21484375,"train/train/layer_model_layers_55/grad/norm":0.1631679017559708,"train/train/tensor_act_model_layers_26_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/norm":0.002458630835620425,"train/train/tensor_grad_model_layers_61_mlp_gate_proj_weight/std":7.119408866265377e-05,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/frac_near_user_limit":0,"train/loss":24.25695343017578,"train/train/tensor_act_model_layers_25_self_attn/max_abs":0.26171875,"train/train/tensor_act_model_layers_44_self_attn_o_proj/std":0.049259785148007024,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_gate_proj_weight/max_abs":0.00130462646484375,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10/norm":866.249273492732,"train/train/tensor_act_model_layers_3_input_layernorm/mean":-0.009723663330078125,"train/train/tensor_act_model_layers_6_self_attn_q_proj/norm":1255.2693234486528,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/max_abs":0.08740234375,"train/train/tensor_act_model_layers_13/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_16_mlp_gate_proj_weight/std":0.00017632218389593765,"train/train/tensor_act_model_layers_64_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/max_abs":0.00093841552734375,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/std":8.65791572219772e-05,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/mean":1.171603798866272e-06,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/mean":0.00017070770263671875,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/std":8.025434981329893e-05,"train/train/tensor_act_model_layers_22_mlp_down_proj/std":0.01586961476140366,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_down_proj/std":0.016113295473829406,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/max_abs":0.08154296875,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp/std":0.016113295473829406,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_49_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/norm":0.002390960224184723,"train/train/layer__model_layers_51/param/max_abs":1,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn_k_proj/mean":-0.0054779052734375,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/norm":3.59375,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/norm":0.22532103206230888,"train/train/tensor_act_model_layers_42_mlp_down_proj/max_abs":0.0859375,"train/train/layer_model_layers_6/grad/norm":0.609532451225864,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_self_attn_v_proj/norm":1362.036064354567,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/mean":-9.939074516296387e-06,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/mean":-1.000240445137024e-06,"train/train/tensor_param_model_layers_43_mlp_gate_proj_weight/mean":-5.155801773071289e-06,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/norm":0.09854968505452935,"train/train/tensor_param_model_layers_0_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_61_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_4/grad/std":0.0008940679432952692,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/norm":2.546875,"train/train/layer__model_layers_8/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/norm":0.7164029486360683,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/mean":1.4924444258213043e-07,"train/train/tensor_act_model_layers_47_mlp_gate_proj/mean":0.0018205642700195312,"train/train/tensor_act_model_layers_56_mlp_down_proj/max_abs":0.08251953125,"train/train/tensor_act_model_layers_56_self_attn_o_proj/max_abs":0.2373046875,"train/train/layer_model_layers_69/act/norm":9220.284782701981,"train/train/tensor_act_model_layers_53_mlp_up_proj/norm":1845.6117326665385,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/norm":0.002845447103311093,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/std":4.3142786704705166e-05,"train/train/tensor_act_model_layers_80_self_attn_v_proj/norm":1305.9522804496078,"train/train/tensor_act_model_layers_39_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/max_abs":0.08203125,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/norm":0.00012914982804156037,"train/train/tensor_act_model_layers_91_mlp_down_proj/mean":-0.00043582916259765625,"train/train/tensor_act_model_layers_8_mlp_down_proj/std":0.016358192547126074,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_k_proj/std":0.22510149091480558,"train/train/tensor_act_model_layers_40_self_attn_q_proj/mean":0.00296783447265625,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/mean":5.184119800105691e-11,"train/train/tensor_param_model_layers_24_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn/std":0.04895110542683219,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_16/act/std":0.4150529748133956,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_10/grad/std":0.0005264988506639635,"train/train/tensor_act_model_layers_60_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/max_abs":4.26173210144043e-06,"train/train/tensor_act_model_layers_68_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_o_proj/mean":0.002811431884765625,"train/train/tensor_act_model_layers_4_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3/max_abs":0.5078125,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/max_abs":0.00469970703125,"train/train/tensor_act_model_layers_42_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_21/grad/norm":0.2808415385777635,"train/train/tensor_act_model_layers_59_mlp_down_proj/std":0.016296647694087132,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/std":0.0004816059331258517,"train/train/tensor_act_model_layers_58_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/max_abs":0.00080108642578125,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_91/grad/std":0.00016982829666084895,"train/train/tensor_param_model_layers_89_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/mean":1.1580996215343475e-06,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/max_abs":0.083984375,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/max_abs":0.005706787109375,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/std":6.83197413293301e-05,"train/train/tensor_act_model_layers_88_mlp_up_proj/norm":1832.9074550075513,"train/train/tensor_grad_model_layers_93_mlp_gate_proj_weight/std":7.290986381089856e-05,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_post_attention_layernorm/mean":0.0102996826171875,"train/train/tensor_param_model_layers_31_mlp_gate_proj_weight/mean":0.00011157989501953125,"train/train/tensor_act_model_layers_4_mlp_gate_proj/norm":1855.9615863368151,"train/train/tensor_act_model_layers_83_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_45/act/max_abs":4.875,"train/train/tensor_act_model_layers_32_mlp_up_proj/mean":-0.00679779052734375,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/max_abs":0.0012359619140625,"train/train/tensor_act_model_layers_5/max_abs":0.62890625,"train/train/tensor_act_model_layers_15_mlp/mean":-0.00047397613525390625,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/max_abs":0.00086212158203125,"train/train/layer_model_layers_56/act/max_abs":4.96875,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/max_abs":0.0013885498046875,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/norm":0.14720039845702934,"train/train/tensor_grad_model_layers_67_mlp_gate_proj_weight/max_abs":0.00096893310546875,"train/train/tensor_act_model_layers_11_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/norm":0.04231369782052575,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/mean":-4.325556801632047e-09,"train/train/tensor_param_model_layers_26_mlp_gate_proj_weight/norm":3.59375,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/max_abs":0.000732421875,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/norm":0.00012048822047452747,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/mean":1.4889519661664963e-07,"train/train/tensor_act_model_layers_44_self_attn_q_proj/max_abs":1.1015625,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/mean":-2.3078173398971558e-06,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/mean":0.00017452239990234375,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/max_abs":0.000396728515625,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/mean":1.3361568562686443e-07,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/std":0.00011975306929253895,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/max_abs":0.0849609375,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_27/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn/mean":0.0023040771484375,"train/train/tensor_act_model_layers_5_self_attn_v_proj/std":0.22315007782652088,"train/train/tensor_act_model_layers_86_self_attn_q_proj/std":0.22559802737185763,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp_up_proj/norm":1861.6839684308222,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/mean":-3.684544935822487e-08,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_input_layernorm/max_abs":4.5,"train/train/layer__model_layers_3/param/max_abs":1,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/norm":0.0011009417219437004,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/max_abs":0.00102996826171875,"train/train/tensor_act_model_layers_2_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer_model_layers_62/grad/std":0.0002145034587417062,"train/train/tensor_act_model_layers_6/std":0.10583718175438955,"train/train/tensor_act_model_layers_55_mlp_up_proj/norm":1850.038736708158,"train/train/tensor_grad_model_layers_79_mlp_gate_proj_weight/norm":0.02365779659501739,"train/train/tensor_act_model_layers_79_post_attention_layernorm/norm":5792.596191409203,"train/train/tensor_act_model_layers_88_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68/std":0.4155361478386299,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/mean":-2.0384788513183594e-05,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_o_proj/norm":294.7954106988249,"train/train/layer__model_layers_29/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/std":0.0004434116568701531,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/std":5.915546988480955e-07,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_mlp_gate_proj_weight/max_abs":0.09130859375,"train/train/layer__model_layers_49/param/frac_near_user_limit":0,"train/train/layer_model_layers_31/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn_k_proj/max_abs":1.125,"train/train/tensor_act_model_layers_88_self_attn/mean":-0.0002859830856323242,"train/train/tensor_act_model_layers_82_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_up_proj/mean":-0.0019702911376953125,"train/train/tensor_act_model_layers_56_mlp_gate_proj/max_abs":1.09375,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/norm":0.006123129799994659,"train/train/tensor_act_model_layers_75_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_post_attention_layernorm/norm":5792.473266607731,"train/train/tensor_grad_model_layers_81_mlp_gate_proj_weight/std":7.416237812473752e-05,"train/train/tensor_act_lm_head/mean":0.0014324188232421875,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/max_abs":9.632110595703125e-05,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/mean":-3.3795833587646484e-05,"train/train/tensor_act_model_layers_12_mlp_gate_proj/mean":0.00396728515625,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp_up_proj/max_abs":1.1328125,"train/train/tensor_act_model_layers_42_self_attn_k_proj/max_abs":1.1484375,"train/train/tensor_param_model_layers_55_self_attn_v_proj_weight/max_abs":0.09521484375,"train/train/tensor_param_model_layers_22_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/max_abs":0.00130462646484375,"train/train/tensor_act_model_layers_46_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_10/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_self_attn_q_proj/std":0.22656632442781677,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_72_mlp_gate_proj/norm":1831.136577037991,"train/train/layer_model_layers_93/grad/max_abs":0.0040283203125,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/mean":-5.1140785217285156e-05,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/std":0.001054043355857016,"train/train/tensor_act_model_layers_59_self_attn/max_abs":0.2255859375,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/mean":-0.00017833709716796875,"train/train/tensor_act_model_layers_39_mlp_gate_proj/norm":1838.7108494467839,"train/train/tensor_act_model_layers_69_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/norm":0.0009007089152101616,"train/train/tensor_act_model_layers_25_self_attn_q_proj/std":0.22387835919220483,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/mean":0.00015735626220703125,"train/train/tensor_act_model_layers_41_self_attn_k_proj/max_abs":1.1015625,"train/train/tensor_grad_model_layers_67_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_75_self_attn/norm":281.61714924777203,"train/train/tensor_act_model_layers_30_mlp/std":0.01669377468626743,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/std":0.0006939353264348596,"train/train/tensor_act_model_layers_83_input_layernorm/norm":5792.593750000072,"train/train/tensor_grad_model_layers_48_mlp_gate_proj_weight/mean":-6.146728992462158e-08,"train/train/tensor_grad_model_layers_12_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_mlp_gate_proj_weight/max_abs":0.08154296875,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_mlp/mean":-0.000316619873046875,"train/train/tensor_act_model_layers_12_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/max_abs":0.08056640625,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/mean":1.2725591659545898e-05,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/max_abs":6.467103958129883e-06,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/norm":0.005569165365813411,"train/train/tensor_act_model_layers_0_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/norm":0.0017237747028616913,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/std":4.901422698937882e-07,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/std":0.0001956113577445048,"train/train/tensor_grad_model_layers_43_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/max_abs":4.082918167114258e-06,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/std":0.0004111410237322948,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/max_abs":0.08056640625,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_88_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp/std":0.014939193218323496,"train/train/tensor_act_model_layers_86_input_layernorm/std":1.000001934442428,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74/std":0.43311312379272343,"train/train/tensor_act_model_layers_73_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_85/act/mean":0.0015488373381750925,"train/train/tensor_act_model_layers_46_mlp_gate_proj/max_abs":1.171875,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_post_attention_layernorm/norm":5792.581542971025,"train/train/tensor_act_model_layers_75_mlp_up_proj/std":0.22583319078697386,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_18_mlp_gate_proj/norm":1872.393450667085,"train/train/layer_model_layers_87/act/mean":0.0034981795719691683,"train/train/tensor_act_model_layers_4/std":0.08142294302093309,"train/train/tensor_act_model_layers_52_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/max_abs":0.0771484375,"train/train/tensor_param_model_layers_56_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp_down_proj/norm":87.62625324793763,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/max_abs":0.004150390625,"train/train/tensor_param_model_embed_tokens_weight/mean":-3.7670135498046875e-05,"train/train/tensor_act_model_layers_60_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_8_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/max_abs":0.083984375,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp/norm":91.49716153139236,"train/train/tensor_param_model_layers_36_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_post_attention_layernorm/norm":5792.575683595283,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/norm":0.0009777929332930044,"train/train/layer_model_layers_69/act/mean":0.0024468153715133667,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/norm":0.0017289376282505573,"train/train/tensor_act_model_layers_72_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/std":9.873125305631637e-05,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/mean":-0.00012493133544921875,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/mean":9.40561294555664e-05,"train/train/tensor_act_model_layers_32_self_attn_k_proj/norm":1275.8571724540755,"train/train/tensor_act_model_layers_72_mlp_gate_proj/std":0.22363717586840445,"train/train/tensor_act_model_layers_30/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/max_abs":0.1015625,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_17_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/std":0.00033562983180840053,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/mean":2.132728695869446e-06,"train/train/tensor_param_model_layers_24_mlp_gate_proj_weight/norm":3.59375,"train/train/tensor_act_model_layers_44_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/mean":-7.224734872579575e-07,"train/train/tensor_grad_model_layers_62_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/std":8.538930767639428e-05,"train/train/tensor_param_model_layers_71_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_input_layernorm/std":1.0000025369194754,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_2/act/max_abs":5.53125,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/max_abs":0.0791015625,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/mean":1,"train/train/global/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_mlp/mean":0.00033974647521972656,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_up_proj/std":0.22778967368266861,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_66_mlp_gate_proj_weight/max_abs":0.0927734375,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_q_proj/max_abs":1.1171875,"train/train/tensor_act_model_layers_29_self_attn_k_proj/norm":1266.8077932884119,"train/train/layer__model_layers_16/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_43_self_attn_v_proj/max_abs":1.078125,"train/train/tensor_act_model_layers_47_post_attention_layernorm/max_abs":4.6875,"train/train/tensor_grad_model_layers_52_mlp_gate_proj_weight/mean":-4.810281097888947e-07,"train/train/tensor_act_model_layers_75_self_attn_k_proj/max_abs":1.1796875,"train/train/tensor_act_model_layers_87_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_embed_tokens_weight/std":0.0008836689374124982,"train/train/tensor_grad_model_layers_60_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_o_proj/max_abs":0.23046875,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/max_abs":4.6193599700927734e-06,"train/train/tensor_grad_model_layers_48_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_72/param/std":0.04425603992050168,"train/train/tensor_act_model_layers_66_self_attn/mean":-0.0017528533935546875,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/mean":-8.58306884765625e-05,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/mean":-0.0001277923583984375,"train/train/tensor_act_model_layers_50_mlp/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/std":7.237054641874154e-05,"train/train/tensor_grad_model_layers_87_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/max_abs":0.0015411376953125,"train/train/tensor_act_model_layers_77_self_attn/mean":-0.0005682706832885742,"train/train/tensor_param_model_layers_55_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/std":0.0004140715600608694,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/norm":0.025513498394021596,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/norm":0.1776471605215164,"train/train/layer__model_layers_23/param/frac_near_user_limit":0,"train/train/tensor_param_model_norm_weight/norm":11.3125,"train/train/tensor_act_model_layers_15_self_attn_v_proj/mean":-0.00482940673828125,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/norm":0.0001714526337787736,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/mean":7.290509529411793e-09,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/mean":6.2659382820129395e-06,"train/train/tensor_act_model_layers_93_mlp/norm":91.69658652261978,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_82_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/mean":1,"train/train/global/param/std":0.04375681321899645,"train/train/tensor_act_model_layers_44_post_attention_layernorm/norm":5792.583129885585,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_89/act/mean":0.003601925713675363,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_14_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_26/grad/std":0.00031578345123830473,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_41/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_post_attention_layernorm/mean":-0.05712890625,"train/train/tensor_param_model_layers_61_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/mean":4.55416738986969e-07,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/std":0.0001234645869739601,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/norm":0.10269182723825086,"train/train/tensor_param_model_layers_69_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/norm":2.59375,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/max_abs":0.0791015625,"train/train/tensor_act_model_layers_5_self_attn_k_proj/max_abs":1.0859375,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/norm":0.00418353172464633,"train/train/tensor_param_model_layers_46_mlp_gate_proj_weight/mean":-0.000125885009765625,"train/train/tensor_act_model_layers_61_post_attention_layernorm/mean":-0.0049936771392822266,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/max_abs":2.0384788513183594e-05,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/max_abs":1.0728836059570312e-05,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/std":5.066489958572901e-07,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/mean":-0.000194549560546875,"train/train/tensor_grad_model_layers_16_mlp_gate_proj_weight/max_abs":0.00244140625,"train/train/tensor_act_model_layers_59_post_attention_layernorm/std":1.0000067660119683,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/mean":-1.0907649993896484e-05,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/std":0.00044258021462719566,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_37/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_mlp_gate_proj_weight/norm":0.03183912249588162,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_10_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp_up_proj/max_abs":1.1953125,"train/train/tensor_act_model_layers_52/max_abs":1.6953125,"train/train/tensor_act_model_layers_22_mlp_up_proj/max_abs":1.296875,"train/train/tensor_act_model_layers_6_self_attn/norm":258.17957838044504,"train/train/layer__model_layers_59/param/norm":17.937302644820235,"train/train/tensor_act_model_layers_23_mlp_down_proj/max_abs":0.08837890625,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/mean":2.7763235266320407e-08,"train/train/tensor_act_model_layers_63_self_attn_v_proj/max_abs":1.3046875,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/std":0.0004257686128815084,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_80_self_attn/max_abs":0.232421875,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/mean":-4.936009645462036e-07,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/std":0.0004562219452185026,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/std":0.00012887496596496496,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_self_attn/std":0.05237228230668631,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/std":7.214411697715757e-05,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/max_abs":0.0260009765625,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_39_input_layernorm/max_abs":4.84375,"train/train/tensor_act_model_layers_41_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn/mean":0.0019741058349609375,"train/train/layer_model_layers_87/act/std":0.42964402051649314,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_60/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn_o_proj/max_abs":0.263671875,"train/train/tensor_act_model_layers_20_self_attn/std":0.049561444386567476,"train/train/tensor_act_model_layers_42_self_attn_o_proj/norm":285.5074063439483,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_77_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/max_abs":0.00024318695068359375,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/max_abs":0.00164031982421875,"train/train/layer_model_layers_45/grad/mean":-2.493107671066305e-07,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_61/grad/norm":0.16676293987621732,"train/train/tensor_grad_model_layers_48_mlp_gate_proj_weight/norm":0.031718849689345775,"train/train/layer_model_layers_57/act/std":0.4225243453837732,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/max_abs":0.09521484375,"train/train/tensor_act_model_layers_75/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/mean":0.00011777877807617188,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/max_abs":0.0791015625,"train/train/layer_model_layers_10/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/mean":-0.00020694732666015625,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_input_layernorm/max_abs":4.75,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_self_attn_o_proj/max_abs":0.259765625,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/max_abs":0.00160980224609375,"train/train/tensor_act_model_layers_84_mlp_gate_proj/std":0.22436999406996117,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/norm":0.0024913916049406027,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_43_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/std":6.309031581627989e-05,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp_up_proj/max_abs":1.125,"train/train/tensor_act_model_layers_56_self_attn_k_proj/std":0.22729704119054817,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn_k_proj/max_abs":0.96484375,"train/train/tensor_act_model_layers_91_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/mean":-4.976755008101463e-09,"train/train/tensor_act_model_layers_68_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/norm":0.00015107913955728254,"train/train/tensor_act_model_layers_17_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56/mean":-0.0016994476318359375,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_42/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_3/param/std":0.044228961362534555,"train/train/tensor_act_model_layers_7_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_64/act/std":0.4248537563711636,"train/train/tensor_act_model_layers_58_self_attn_k_proj/max_abs":1.0390625,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp_down_proj/std":0.01590006214206699,"train/train/tensor_param_model_layers_80_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_19/std":0.21387745775968214,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_mlp_up_proj/max_abs":1.1953125,"train/train/tensor_act_model_layers_55_mlp_gate_proj/norm":1809.5395304526544,"train/train/layer__model_layers_73/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn/std":0.04938035833751627,"train/train/tensor_act_model_layers_32_mlp_down_proj/std":0.015274157458811454,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/max_abs":0.0001697540283203125,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/mean":0.0001506805419921875,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/max_abs":0.0849609375,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn/std":0.051271675674611376,"train/train/tensor_act_model_layers_63_mlp/std":0.01577854052953808,"train/train/tensor_grad_model_embed_tokens_weight/norm":1.2801002367826269,"train/train/layer__model_layers_9/param/std":0.04423413573947681,"train/train/tensor_act_model_layers_27_mlp_up_proj/mean":-0.01019287109375,"train/train/layer__model_layers_32/param/norm":17.931013363286944,"train/train/tensor_param_model_layers_34_mlp_gate_proj_weight/max_abs":0.08203125,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp/mean":8.672475814819336e-05,"train/train/tensor_act_model_layers_79_self_attn_v_proj/max_abs":1.0859375,"train/train/tensor_act_model_layers_89_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_post_attention_layernorm/mean":-0.0067138671875,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_79_mlp_gate_proj_weight/mean":-3.434251993894577e-07,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_28/std":0.2612360851183211,"train/train/tensor_param_model_layers_88_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_81_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/mean":-0.00013446807861328125,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_input_layernorm/max_abs":4.84375,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/std":6.465476702738302e-07,"train/train/layer_model_layers_78/grad/norm":0.15304718312590687,"train/train/tensor_act_model_layers_31_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_15/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_9_mlp_gate_proj_weight/max_abs":0.09521484375,"train/train/tensor_act_model_layers_74_input_layernorm/std":1.0000020789863118,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/std":0.00011259789083920421,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_up_proj/norm":1834.9859804355983,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/std":9.318592075200633e-05,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/max_abs":0.09228515625,"train/train/tensor_act_model_layers_70_mlp_up_proj/max_abs":1.0625,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/std":8.613312618521478e-05,"train/train/tensor_act_model_layers_80_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_54_self_attn_k_proj/mean":-0.0120086669921875,"train/train/tensor_act_model_layers_54_mlp_down_proj/std":0.01568665504245393,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/mean":-2.343207597732544e-06,"train/train/tensor_param_model_layers_60_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/mean":-5.103647708892822e-07,"train/train/tensor_act_model_layers_80_post_attention_layernorm/mean":0.01104736328125,"train/train/tensor_param_model_layers_25_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_v_proj/max_abs":1.03125,"train/train/tensor_act_model_layers_60_self_attn_q_proj/mean":0.0049591064453125,"train/train/tensor_act_model_layers_37/norm":1740.5873299549635,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp_gate_proj/max_abs":1.1796875,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/std":3.7820430500622665e-05,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/norm":0.002185818204494425,"train/train/tensor_param_model_layers_77_mlp_gate_proj_weight/mean":0.000148773193359375,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/max_abs":0.01068115234375,"train/train/tensor_act_model_layers_55_mlp_gate_proj/max_abs":1.125,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/max_abs":0.08154296875,"train/train/tensor_act_model_layers_88_post_attention_layernorm/norm":5792.603759766866,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_34_self_attn_v_proj/mean":-0.0064849853515625,"train/train/tensor_act_model_layers_73_self_attn_o_proj/mean":-0.0006604194641113281,"train/train/tensor_act_model_layers_44_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/mean":-2.050910552497953e-09,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/norm":0.00013240712686653393,"train/train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_v_proj/std":0.22852123031506885,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/norm":0.00015344009245018184,"train/train/tensor_act_model_layers_30_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/mean":3.1905074138194323e-09,"train/train/layer__model_layers_91/param/norm":17.933790720574805,"train/train/tensor_act_model_layers_82_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_k_proj_weight/max_abs":7.539987564086914e-06,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26/norm":1443.8752569398946,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/max_abs":4.947185516357422e-06,"train/train/tensor_act_model_layers_36/mean":-0.001922607421875,"train/train/layer_model_layers_69/grad/max_abs":0.005035400390625,"train/train/layer_model_layers_47/grad/mean":-3.366131323147071e-07,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_64_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/std":6.309927194021005e-07,"train/train/tensor_act_model_layers_75_self_attn_o_proj/std":0.04864668740597001,"train/train/tensor_act_model_layers_93_input_layernorm/mean":0.01544189453125,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/max_abs":0.000942230224609375,"train/train/tensor_act_model_layers_13_input_layernorm/norm":5792.5041503915145,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/max_abs":7.724761962890625e-05,"train/train/tensor_param_model_layers_4_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_33_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_o_proj/std":0.04553422607066122,"train/train/tensor_act_model_layers_56_mlp_gate_proj/norm":1832.5984397819354,"train/train/tensor_param_model_layers_31_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_26/grad/max_abs":0.006103515625,"train/train/tensor_param_model_layers_21_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_input_layernorm/norm":5792.590698243858,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/std":3.1178375521303355e-05,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/mean":3.080815076828003e-06,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/norm":0.06907626675063111,"train/train/tensor_act_model_layers_14_mlp_down_proj/std":0.015839274071041885,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/norm":0.20719759034777133,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_33/grad/max_abs":0.006500244140625,"train/train/tensor_act_model_layers_14_mlp/std":0.015839274071041885,"train/train/tensor_act_model_layers_30_mlp_down_proj/mean":-0.001148223876953125,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/mean":1,"train/train/layer__model_layers_34/param/norm":17.93507714473233,"train/train/tensor_act_model_layers_61_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_mlp_gate_proj_weight/max_abs":0.00091552734375,"train/train/tensor_act_model_layers_30_mlp_up_proj/std":0.23046921234594658,"train/train/tensor_param_model_layers_90_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/std":0.02001953125,"train/train/time_per_step_avg":1.5960035105235875,"train/train/tensor_param_model_layers_85_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/max_abs":0.0062255859375,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/mean":9.1552734375e-05,"train/train/tensor_act_model_layers_87/norm":2722.1878742711538,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/mean":-3.409385681152344e-05,"train/train/layer__model_layers_53/param/norm":17.927500206996932,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/std":0.0201416015625,"train/train/layer_model_layers_74/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/std":0.0003413368840999018,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/std":4.2996047522089444e-07,"train/train/tensor_act_model_layers_30_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp_down_proj/norm":89.33651009631114,"train/train/tensor_act_model_layers_76_mlp_gate_proj/norm":1804.6819127631975,"train/train/layer_model_layers_50/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_43/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_56/grad/norm":0.169832464202999,"train/train/tensor_act_model_layers_76_mlp_up_proj/max_abs":1.078125,"train/train/tensor_param_model_layers_13_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_79_input_layernorm/std":1.0000020612093652,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/mean":-2.3365020751953125e-05,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/mean":6.41215592622757e-07,"train/train/tensor_act_model_layers_86_self_attn_o_proj/std":0.048525337544329854,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/mean":6.103515625e-05,"train/train/tensor_act_model_layers_46_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp/max_abs":0.08837890625,"train/train/tensor_act_model_layers_71_mlp_up_proj/max_abs":1.09375,"train/train/tensor_param_model_layers_70_mlp_gate_proj_weight/mean":0.00022220611572265625,"train/train/layer__model_layers_54/param/std":0.044219042890243296,"train/train/tensor_grad_model_layers_86_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/max_abs":0.09130859375,"train/train/tensor_act_model_layers_45_mlp_up_proj/std":0.22803071119611543,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/norm":0.00016036410076340613,"train/train/tensor_act_model_layers_65_self_attn_v_proj/norm":1338.94020864181,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/max_abs":0.00125885009765625,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/max_abs":0.004180908203125,"train/train/tensor_act_model_layers_59_mlp_gate_proj/mean":-0.0045013427734375,"train/train/tensor_act_model_layers_36_mlp_up_proj/frac_near_user_limit":0,"train/train/layer_model_layers_6/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_mlp_gate_proj_weight/max_abs":0.0810546875,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/mean":5.033143679611385e-09,"train/train/tensor_act_model_layers_28_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_q_proj/max_abs":1.078125,"train/train/tensor_act_model_layers_56/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_q_proj/mean":-0.0027818679809570312,"train/train/tensor_act_model_layers_84_self_attn_k_proj/max_abs":1.09375,"train/train/tensor_param_model_layers_79_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_88/act/std":0.42976763004727786,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/std":0.000529787179277424,"train/train/layer_model_layers_93/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn_o_proj/mean":-0.00015676021575927734,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/norm":0.0012158954838414075,"train/train/tensor_act_model_layers_83_input_layernorm/mean":0.011871337890625,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88/mean":0.004726409912109375,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_78_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27/std":0.2539106382394265,"train/train/layer_model_layers_71/grad/frac_near_user_limit":0,"train/train/layer_model_layers_84/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37/max_abs":1.5078125,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_52_mlp_down_proj/std":0.015488088323518473,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/norm":2.53125,"train/train/tensor_act_model_layers_4_self_attn/norm":254.09113697599594,"train/train/tensor_param_model_layers_49_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/std":8.986891205499497e-07,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_88/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/max_abs":0.08203125,"train/train/tensor_act_model_layers_42/std":0.3178765386929583,"train/train/tensor_act_model_layers_30_input_layernorm/max_abs":4.75,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_46/std":0.3335013735995072,"train/train/tensor_act_model_layers_12_mlp_up_proj/norm":1843.231572138121,"train/train/layer_model_layers_65/grad/norm":0.17084840128278056,"train/train/tensor_act_model_layers_19_self_attn_v_proj/norm":1344.438559978378,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_6_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_o_proj/norm":288.78435134613994,"train/train/tensor_act_model_layers_80_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/max_abs":3.9637088775634766e-06,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_68_mlp_down_proj/mean":-9.28640365600586e-05,"train/train/tensor_act_model_layers_12/norm":955.2431172466103,"train/train/tensor_param_model_layers_83_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/max_abs":4.4375,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/std":0.00020098475247181562,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/max_abs":0.00072479248046875,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/max_abs":0.0169677734375,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_25_self_attn_k_proj/mean":-0.012237548828125,"train/train/tensor_act_model_layers_90_mlp/max_abs":0.0849609375,"train/train/tensor_param_model_layers_54_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/std":0.0005368207770062621,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_91_self_attn_v_proj/mean":-0.00017452239990234375,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/std":0.00010394449578774051,"train/train/tensor_act_model_layers_73_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_20/grad/mean":-1.5377455141087311e-06,"train/train/tensor_act_model_layers_18_post_attention_layernorm/mean":-0.05352783203125,"train/train/tensor_param_model_layers_87_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/mean":2.8014183044433594e-05,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/max_abs":6.467103958129883e-06,"train/train/tensor_grad_model_layers_17_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_v_proj/std":0.22803087127747615,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp_down_proj/norm":90.17691290225004,"train/train/tensor_param_model_layers_2_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_mlp_down_proj_weight/std":6.880226339947925e-05,"train/train/tensor_act_model_layers_55_mlp_down_proj/norm":86.69915456850266,"train/train/layer__model_layers_29/param/mean":0.001630744398477111,"train/train/tensor_act_model_layers_9_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/mean":2.586841583251953e-05,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/max_abs":0.00139617919921875,"train/train/tensor_act_model_layers_22_self_attn/std":0.04705999858761875,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/std":8.303610291624284e-05,"train/train/tensor_act_model_layers_21_self_attn/norm":269.06969488437073,"train/train/tensor_act_model_layers_71_mlp/frac_near_user_limit":0,"train/train/layer__model_layers_28/param/max_abs":1,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/max_abs":0.00022220611572265625,"train/train/tensor_act_model_layers_26_mlp_up_proj/frac_near_user_limit":0,"train/train/layer_model_layers_28/act/mean":-0.00895294121333531,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/max_abs":0.00030517578125,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp_gate_proj/mean":-0.004650115966796875,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/norm":0.6141887195342528,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/mean":0.00014209747314453125,"train/train/tensor_act_model_layers_44_self_attn_q_proj/std":0.22900791764558678,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/max_abs":0.00433349609375,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/mean":-4.0531158447265625e-06,"train/train/tensor_param_model_layers_35_mlp_gate_proj_weight/max_abs":0.08203125,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/std":3.6030261843956564e-05,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/std":0.00012884848124075329,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/max_abs":0.00022602081298828125,"train/train/layer_model_layers_6/grad/frac_near_user_limit":0,"train/train/layer_model_layers_80/act/max_abs":4.75,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_input_layernorm/std":1.000004788849212,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77/norm":2560.0755972875345,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_mlp_gate_proj_weight/std":0.00010955694058803155,"train/train/tensor_param_model_layers_15_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_37_self_attn_o_proj/max_abs":0.228515625,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_gate_proj_weight/max_abs":0.0017547607421875,"train/train/tensor_act_model_layers_49_self_attn/norm":287.8277077107872,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/mean":-0.00011444091796875,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_46_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/mean":4.076957702636719e-05,"train/train/tensor_act_model_layers_13_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/std":7.20284663457655e-05,"train/train/layer_model_layers_7/grad/std":0.0006687137321184786,"train/train/tensor_param_model_layers_41_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/max_abs":0.0010986328125,"train/train/tensor_param_model_layers_92_input_layernorm_weight/mean":1,"train/train/layer_model_layers_32/grad/frac_near_user_limit":0,"train/train/layer__model_layers_47/param/mean":0.001562079847695861,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_48_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn_k_proj/std":0.23560819101714633,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/std":8.92260121186777e-07,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/max_abs":0.08203125,"train/train/tensor_act_model_layers_32_self_attn_q_proj/norm":1298.0025082823688,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_67/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp_gate_proj/std":0.2231478117336163,"train/train/tensor_param_model_layers_85_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_55_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_mlp_gate_proj_weight/std":7.213456325613677e-05,"train/train/tensor_param_model_layers_26_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_83/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_input_layernorm/max_abs":4.84375,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_87/param/max_abs":1,"train/train/tensor_act_model_layers_47_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/max_abs":0.07861328125,"train/train/tensor_param_model_layers_89_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/max_abs":0.08837890625,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn_q_proj/mean":-0.00745391845703125,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/mean":3.203749656677246e-06,"train/train/tensor_act_model_layers_46_post_attention_layernorm/mean":-0.002780914306640625,"train/train/layer_model_layers_76/grad/norm":0.1565388033477511,"train/train/tensor_act_model_layers_40/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_q_proj/max_abs":1.0234375,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/max_abs":0.080078125,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/std":0.00033753214451738334,"train/train/tensor_act_model_layers_76_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer__model_layers_3/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp/max_abs":0.0830078125,"train/train/tensor_grad_model_layers_42_mlp_gate_proj_weight/max_abs":0.000949859619140625,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/max_abs":0.0113525390625,"train/train/tensor_act_model_layers_45_self_attn_q_proj/mean":0.00392913818359375,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/max_abs":0.000888824462890625,"train/train/tensor_act_model_layers_32_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_gate_proj/mean":-0.0093994140625,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_43/grad/std":0.00022769241473988874,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/mean":-3.528594970703125e-05,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_75/grad/norm":0.15107010162795784,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/norm":0.0013033319486752864,"train/train/tensor_act_model_layers_37_self_attn_q_proj/mean":-0.004611968994140625,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_9_post_attention_layernorm/max_abs":4.84375,"train/train/layer_model_layers_44/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp_gate_proj/norm":1872.2629483774492,"train/train/tensor_act_model_layers_76_mlp_down_proj/norm":86.67957611304256,"train/train/tensor_param_model_layers_6_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62/std":0.392588526089342,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/std":0.00012260268266391412,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/max_abs":0.0003490447998046875,"train/train/tensor_act_model_layers_29_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_31_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_o_proj/mean":-0.002841949462890625,"train/train/tensor_act_model_layers_44_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_90/act/mean":0.0026084695543561664,"train/train/tensor_act_model_layers_24_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_79_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_mlp_up_proj/mean":-0.00618743896484375,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_86_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_input_layernorm/max_abs":4.71875,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/mean":1.341104507446289e-05,"train/train/tensor_act_model_layers_58_self_attn_k_proj/std":0.22583518542816458,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/max_abs":0.08203125,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/max_abs":0.00116729736328125,"train/train/tensor_act_model_layers_61_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/max_abs":0.08056640625,"train/train/tensor_act_model_layers_62_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/max_abs":0.0830078125,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/mean":-8.916854858398438e-05,"train/train/tensor_act_model_layers_78_mlp/mean":2.0682811737060547e-05,"train/train/tensor_param_model_layers_43_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp/std":0.01556396670046014,"train/train/tensor_act_model_layers_41_self_attn_v_proj/norm":1309.9687408698705,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/max_abs":0.0859375,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp_down_proj/norm":88.81283999801195,"train/train/tensor_act_model_layers_14_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/mean":7.137714419513941e-08,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn_o_proj/std":0.04956111740084291,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/mean":1.21421180665493e-07,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_q_proj/max_abs":1.1640625,"train/train/tensor_act_model_layers_54/frac_near_user_limit":0,"train/train/layer_model_layers_42/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_67/grad/norm":0.16630268923601105,"train/train/tensor_act_model_layers_85_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_89/param/norm":17.94277334289142,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_2_mlp_down_proj/max_abs":0.09423828125,"train/train/tensor_grad_model_layers_48_mlp_gate_proj_weight/max_abs":0.001312255859375,"train/train/tensor_act_model_layers_62_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_post_attention_layernorm/max_abs":4.71875,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_input_layernorm_weight/norm":11.3125,"train/train/layer__model_layers_18/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_self_attn_q_proj/norm":1328.7599380010372,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_10/param/max_abs":1,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/max_abs":8.64267349243164e-06,"train/train/tensor_act_model_layers_47_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/max_abs":0.07861328125,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/mean":-1.0691583156585693e-05,"train/train/tensor_act_model_layers_92_self_attn/std":0.05078515950585719,"train/train/tensor_act_model_layers_81_self_attn_v_proj/std":0.2287657393700445,"train/train/tensor_act_model_layers_65_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_mlp_gate_proj_weight/std":0.02001953125,"train/train/layer__model_layers_73/param/norm":17.92782023228842,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_38_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/norm":0.056748517634655216,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/max_abs":6.139278411865234e-06,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/mean":-9.822845458984375e-05,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_49/grad/norm":0.17856196368505345,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_norm_weight/std":0.0010189541478822918,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/max_abs":0.000148773193359375,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/max_abs":5.125999450683594e-06,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/mean":6.298068910837173e-08,"train/train/tensor_act_model_layers_36_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/max_abs":0.08984375,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/norm":0.08704186804495366,"train/train/layer_model_layers_76/act/std":0.42659799964099987,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/mean":-5.14984130859375e-05,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp_gate_proj/norm":1849.0976876020804,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_83_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/std":7.586220588132905e-07,"train/train/tensor_param_model_layers_53_mlp_gate_proj_weight/std":0.0198974609375,"train/train/layer_model_layers_69/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/mean":5.990266799926758e-06,"train/train/tensor_act_model_layers_11_self_attn_o_proj/max_abs":0.236328125,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_k_proj/std":0.22095522945235563,"train/train/tensor_act_model_layers_14_self_attn/mean":-0.0006768703460693359,"train/train/tensor_act_model_layers_41_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_o_proj/norm":287.8277077107872,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_post_attention_layernorm/max_abs":5.0625,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/max_abs":0.0859375,"train/train/tensor_act_model_layers_32_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_gate_proj_weight/mean":-4.5299530029296875e-05,"train/train/layer__model_layers_87/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/mean":7.286667823791504e-06,"train/train/tensor_act_model_layers_77_mlp_up_proj/max_abs":1.171875,"train/train/tensor_act_model_layers_89_mlp/mean":0.0006132125854492188,"train/train/layer_model_layers_4/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_80/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/mean":0.000202178955078125,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/max_abs":4.678964614868164e-06,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/mean":0.00019168853759765625,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/max_abs":4.798173904418945e-06,"train/train/tensor_param_model_layers_15_mlp_gate_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_59_self_attn/std":0.0496236921594323,"train/train/tensor_act_model_layers_50/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/max_abs":0.0010986328125,"train/train/tensor_act_model_layers_79/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_k_proj/max_abs":1.1328125,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_q_proj/max_abs":1.0859375,"train/train/layer_model_layers_44/grad/mean":2.2822461060470612e-07,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13/norm":995.9246147512089,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_post_attention_layernorm/norm":5792.586303714965,"train/train/tensor_act_model_layers_55_mlp/max_abs":0.07861328125,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_1_mlp_up_proj/max_abs":1.25,"train/train/tensor_act_model_layers_52_mlp_gate_proj/mean":4.863739013671875e-05,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_mlp_down_proj/mean":-0.0001482069492340088,"train/train/tensor_act_model_layers_52_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_k_proj/std":0.22339125824561906,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_down_proj/std":0.01599240766133549,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_input_layernorm/std":0.9990329145353037,"train/train/tensor_act_model_layers_90_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/std":0.020263671875,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_61_mlp_gate_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/mean":-5.125999450683594e-06,"train/train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/std":4.211830733259999e-07,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/max_abs":0.00177764892578125,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_31_self_attn/std":0.04742573627196074,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_78_mlp_gate_proj/max_abs":1.125,"train/train/layer__model_layers_76/param/mean":0.0015591608008802774,"train/train/layer__model_layers_55/param/max_abs":1,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_90_mlp_up_proj/mean":0.0005383491516113281,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_o_proj/norm":284.3693547447153,"train/train/tensor_param_model_layers_89_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_24/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/std":0.0004005898299632689,"train/train/tensor_act_model_layers_21_mlp_down_proj/std":0.015458980657959814,"train/train/tensor_act_model_layers_65_self_attn_v_proj/max_abs":1.1484375,"train/train/layer__model_layers_52/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp_up_proj/max_abs":1.21875,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/std":3.386856576768815e-05,"train/train/tensor_act_model_layers_26_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_19/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/norm":0.028632059542759505,"train/train/tensor_act_model_layers_40_self_attn_o_proj/mean":0.001842498779296875,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/max_abs":0.0037384033203125,"train/train/tensor_act_model_layers_48_self_attn_k_proj/max_abs":1.0859375,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_89/param/std":0.04426273167789703,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/max_abs":0.0966796875,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/mean":-3.3651303965598345e-10,"train/train/tensor_act_model_layers_27_self_attn_q_proj/mean":-0.00090789794921875,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/norm":0.00010453016626917915,"train/train/tensor_act_model_layers_73_post_attention_layernorm/std":1.0000017822205827,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_44_mlp_up_proj/std":0.2292501201106788,"train/train/tensor_param_model_layers_52_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_51_mlp_gate_proj_weight/mean":-1.1164229363203049e-07,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/mean":-1.0442454367876053e-07,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/max_abs":0.00019550323486328125,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/max_abs":0.08447265625,"train/train/tensor_act_model_layers_41_post_attention_layernorm/max_abs":4.96875,"train/train/tensor_act_model_layers_83_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/std":0.00016017280707908253,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/norm":0.00160791186666528,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_62_mlp_gate_proj_weight/max_abs":0.00104522705078125,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/std":0.0004177496681475185,"train/train/tensor_act_model_layers_58/max_abs":1.84375,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/norm":0.002904235029171364,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_22_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_k_proj/std":0.22949505714435597,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/mean":2.332031726837158e-06,"train/train/layer_model_layers_19/act/norm":8990.193874352582,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn_k_proj/mean":-0.004756927490234375,"train/train/tensor_act_model_layers_50_self_attn_q_proj/std":0.22583333605916445,"train/train/layer_model_layers_9/act/std":0.4124369260123285,"train/train/tensor_act_model_layers_85_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/norm":0.025855503656848168,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/std":0.0008030796899313785,"train/train/tensor_act_model_layers_60_self_attn/norm":285.08397855122234,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/mean":3.001332515850663e-10,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/std":4.749144546554289e-05,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/max_abs":0.001983642578125,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_54_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/max_abs":0.08251953125,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/mean":-0.00011348724365234375,"train/train/tensor_act_model_layers_80_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/norm":0.00017448917447860767,"train/train/tensor_act_model_layers_40_self_attn_k_proj/std":0.23438164702465358,"train/train/tensor_param_model_layers_8_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_23/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/norm":0.10191752663657892,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/mean":-0.0001544952392578125,"train/train/tensor_act_model_layers_8_input_layernorm/std":0.996101349446348,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/mean":5.461042746901512e-07,"train/train/tensor_act_model_layers_91_post_attention_layernorm/max_abs":4.5625,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_53/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/mean":-1.389533281326294e-06,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/norm":0.006125600911850006,"train/train/tensor_act_model_layers_33_mlp/std":0.014984534313884093,"train/train/tensor_act_model_layers_89_self_attn/norm":274.62135541157573,"train/train/tensor_act_model_layers_23_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18/max_abs":0.95703125,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/norm":0.027659365807142657,"train/train/tensor_act_model_layers_86_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/norm":0.0001109699548901864,"train/train/tensor_act_model_layers_86_mlp_down_proj/mean":-0.00017890334129333496,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/mean":-3.0850060284137726e-09,"train/train/tensor_act_model_layers_23_mlp_up_proj/mean":-0.0028533935546875,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_mlp_gate_proj_weight/max_abs":0.08642578125,"train/train/tensor_act_model_layers_26_mlp/norm":89.78689562465455,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/mean":-1.8384307622909546e-06,"train/train/layer_model_layers_1/act/norm":8892.08960213552,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_64/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_k_proj/mean":-0.009429931640625,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_v_proj/std":0.22950176906367295,"train/train/tensor_act_model_layers_31/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_post_attention_layernorm/norm":5792.529296878077,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/norm":0.00030597492726757846,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/mean":9.480572771281004e-09,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/max_abs":2.3484230041503906e-05,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/norm":0.43877443251004467,"train/train/tensor_param_model_layers_62_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_62_mlp_gate_proj/std":0.22388185508211952,"train/train/layer_model_layers_62/act/mean":0.0001448350293295724,"train/train/layer__model_layers_58/param/std":0.04425258087799161,"train/train/tensor_param_model_layers_89_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/max_abs":0.0013885498046875,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/mean":0.00013637542724609375,"train/train/tensor_act_model_layers_28_self_attn_o_proj/norm":293.8350357628902,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/std":0.0005079424272055849,"train/train/tensor_grad_model_layers_35_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_2/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_72/grad/max_abs":0.00506591796875,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/mean":-0.0001926422119140625,"train/train/tensor_act_model_layers_53/max_abs":1.75,"train/train/tensor_act_model_layers_38_mlp_gate_proj/max_abs":1.3046875,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_mlp_gate_proj_weight/norm":0.030030165248813787,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/mean":-0.00013828277587890625,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/mean":-7.767230272293091e-07,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/norm":0.0018616621589843975,"train/train/tensor_grad_model_layers_82_mlp_gate_proj_weight/max_abs":0.00086212158203125,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_v_proj/std":0.22778543074986407,"train/train/tensor_act_model_layers_31_mlp/max_abs":0.08544921875,"train/train/tensor_param_model_layers_37_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_o_proj/max_abs":0.2421875,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/norm":0.10694390512856133,"train/train/tensor_act_model_layers_84_self_attn_k_proj/norm":1362.6898543239872,"train/train/tensor_act_model_layers_47_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/max_abs":0.000965118408203125,"train/train/tensor_act_model_layers_23_self_attn_k_proj/max_abs":1.09375,"train/train/tensor_act_model_layers_32_self_attn_o_proj/max_abs":0.2275390625,"train/train/layer__model_layers_1/param/std":0.04423040208084078,"train/train/tensor_act_model_layers_52_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_75/max_abs":2.15625,"train/train/layer_model_layers_63/grad/max_abs":0.005096435546875,"train/train/layer__model_layers_77/param/std":0.04423661286091243,"train/epoch":0.021569156106767323,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/max_abs":0.00171661376953125,"train/train/layer_model_layers_38/act/mean":-0.004516533442905971,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/norm":0.0025026856279991823,"train/train/tensor_act_model_layers_35/norm":1691.4529869237417,"train/train/tensor_act_model_layers_54/mean":-0.0038814544677734375,"train/train/layer_model_layers_36/grad/std":0.0002768984992858009,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/norm":9.280701589570194e-05,"train/train/tensor_grad_model_layers_74_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_q_proj/max_abs":1.1875,"train/train/tensor_act_model_layers_89_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn/norm":270.34992872621643,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/mean":-2.091837814077735e-09,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/mean":-0.00010061264038085938,"train/train/tensor_act_model_layers_38_self_attn_q_proj/max_abs":1.1015625,"train/train/tensor_act_model_layers_23_self_attn_v_proj/std":0.23316180668964454,"train/train/tensor_act_model_layers_2_mlp_down_proj/mean":-0.0002238750457763672,"train/train/tensor_grad_model_layers_0_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_72/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_76/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_mlp_gate_proj_weight/max_abs":0.0015106201171875,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86/norm":2720.582604358494,"train/train/tensor_act_model_layers_64_input_layernorm/std":1.0000084833813485,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/std":0.00015260045786906354,"train/train/tensor_act_model_layers_59/std":0.38135543268320526,"train/train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_q_proj/norm":1318.1140529263512,"train/train/tensor_param_model_layers_41_mlp_gate_proj_weight/norm":3.59375,"train/train/tensor_grad_model_layers_70_mlp_gate_proj_weight/norm":0.02618898999293067,"train/train/tensor_act_model_layers_34_mlp_down_proj/mean":-0.0004787445068359375,"train/train/tensor_act_model_layers_16_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43/mean":-0.004390716552734375,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/std":0.0003708596584545054,"train/train/tensor_act_model_layers_47/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/std":0.00039811666430671257,"train/train/tensor_act_model_layers_32_input_layernorm/max_abs":4.9375,"train/train/tensor_act_model_layers_30_input_layernorm/norm":5792.569946289581,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/max_abs":0.0011444091796875,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/mean":2.6728957891464233e-06,"train/train/layer_model_layers_10/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_q_proj/norm":1280.5896643250364,"train/train/tensor_act_model_layers_45_mlp_up_proj/norm":1868.6381171983453,"train/train/layer_model_layers_79/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/mean":7.366761565208435e-07,"train/train/tensor_act_model_layers_54_mlp/max_abs":0.08837890625,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/mean":-4.506111145019531e-05,"train/train/tensor_act_model_layers_3_mlp_down_proj/mean":-0.0007886886596679688,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/max_abs":0.003509521484375,"train/train/tensor_act_model_layers_65_mlp_gate_proj/mean":0.00269317626953125,"train/train/tensor_act_model_layers_92_self_attn_k_proj/std":0.23169092835096358,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41/frac_near_dtype_limit":0,"train/train/layer__model_layers_91/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_85_self_attn_o_proj/norm":291.14040547445387,"train/train/tensor_act_model_layers_82_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/max_abs":0.09326171875,"train/train/tensor_act_model_layers_20_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/max_abs":0.09619140625,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/mean":-0.00015163421630859375,"train/train/tensor_act_model_layers_44_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_12_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/norm":0.0006899041101980986,"train/train/tensor_act_model_layers_36_self_attn_o_proj/mean":0.0023040771484375,"train/train/tensor_act_model_layers_66_self_attn_k_proj/max_abs":1.0859375,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/mean":6.421469151973724e-07,"train/train/tensor_act_model_layers_47_mlp_up_proj/max_abs":1.0625,"train/train/tensor_act_model_layers_93_mlp_down_proj/std":0.015839508765506258,"train/train/tensor_act_model_layers_84_post_attention_layernorm/std":1.000000973697278,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/norm":0.13582822419525742,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp/mean":0.00017213821411132812,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/norm":0.0003986860544832878,"train/train/layer_model_layers_80/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/max_abs":0.07470703125,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/mean":-4.526227712631226e-06,"train/train/tensor_act_model_layers_27_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_self_attn_v_proj/max_abs":1.234375,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/max_abs":0.0869140625,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/max_abs":0.009521484375,"train/train/tensor_act_model_layers_52_mlp_up_proj/max_abs":1.1015625,"train/train/tensor_act_model_layers_74_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_input_layernorm/std":1.0000051090697981,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/std":0.0002530313420908181,"train/train/tensor_act_model_layers_16/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/norm":0.09760649367211237,"train/train/tensor_act_model_layers_31/mean":-0.00823974609375,"train/train/tensor_act_model_layers_84_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_86/act/std":0.4296307589074943,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/std":0.000402838998808672,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_53/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn_k_proj/std":0.2258317792673734,"train/train/tensor_act_model_layers_55_self_attn_v_proj/std":0.22461292442855113,"train/train/tensor_act_model_layers_84_mlp_up_proj/std":0.22729990364191868,"train/train/tensor_act_model_layers_30_mlp/norm":96.91723356035952,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_gate_proj_weight/norm":0.0298577346790363,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/std":6.0075727732595407e-05,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/std":0.0004888821011635089,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/max_abs":0.0849609375,"train/train/tensor_act_model_layers_33_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/mean":0.0001983642578125,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_59_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57/std":0.372570183489798,"train/train/tensor_act_model_layers_58_mlp_up_proj/std":0.2258337186829211,"train/train/tensor_act_model_layers_71_self_attn_o_proj/mean":0.0004916191101074219,"train/train/tensor_act_model_layers_18_self_attn_o_proj/norm":261.8512156003335,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_60_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_o_proj/max_abs":0.251953125,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/max_abs":0.003448486328125,"train/train/tensor_act_model_layers_14_post_attention_layernorm/max_abs":5.0625,"train/train/tensor_act_model_layers_51_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_post_attention_layernorm/max_abs":4.8125,"train/train/tensor_grad_model_layers_81_mlp_gate_proj_weight/mean":-5.5588316172361374e-08,"train/train/layer__model_layers_51/param/norm":17.938629646484983,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/norm":0.001093288509471688,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/mean":5.173683166503906e-05,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_14_self_attn_k_proj/mean":0.004268646240234375,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/max_abs":0.006072998046875,"train/train/tensor_act_model_layers_66_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_28/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp/std":0.015732491614499727,"train/train/tensor_grad_model_layers_10_mlp_gate_proj_weight/norm":0.07332596120480717,"train/train/tensor_act_model_layers_35_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_post_attention_layernorm/std":0.9990318704437493,"train/train/tensor_act_model_layers_3_self_attn_k_proj/max_abs":1.265625,"train/train/tensor_act_model_layers_13_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/max_abs":0.08544921875,"train/train/tensor_act_model_layers_58_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_gate_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_66_post_attention_layernorm/norm":5792.591674807737,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/max_abs":0.08154296875,"train/train/tensor_param_model_layers_82_mlp_gate_proj_weight/norm":3.59375,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/std":0.02001953125,"train/train/layer__model_layers_1/param/mean":0.001564588264072555,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/norm":0.187953400781892,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/mean":4.32133674621582e-06,"train/train/tensor_param_model_layers_82_mlp_gate_proj_weight/std":0.019775390625,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_k_proj/norm":1345.5868629156346,"train/train/layer_model_layers_8/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/max_abs":0.0003261566162109375,"train/train/layer__model_layers_89/param/max_abs":1,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/norm":0.06693283432373016,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/mean":4.982948303222656e-05,"train/train/tensor_act_model_layers_3_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/max_abs":0.00177001953125,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_up_proj/max_abs":1.0625,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/std":7.793290779176481e-05,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/std":7.953575273281674e-05,"train/train/tensor_param_model_layers_85_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_post_attention_layernorm/norm":5792.591186524841,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_mlp_gate_proj_weight/norm":0.04236648878069694,"train/train/tensor_act_model_layers_73_self_attn_o_proj/std":0.04889374723608988,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/std":1.1448851224950906e-06,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_35_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/mean":0.00010251998901367188,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_44_input_layernorm/std":1.0000046691748272,"train/train/tensor_param_model_layers_40_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_89_self_attn_v_proj/std":0.21973154416662416,"train/train/layer_model_layers_55/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/mean":-6.552785634994507e-06,"train/train/layer_model_layers_86/act/mean":0.00486946531704494,"train/train/tensor_act_model_layers_13_input_layernorm/max_abs":4.84375,"train/train/tensor_grad_model_layers_30_mlp_gate_proj_weight/max_abs":0.0016021728515625,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/max_abs":0.08544921875,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_29_self_attn_v_proj/mean":-0.0019121170043945312,"train/train/tensor_act_model_layers_68_self_attn/std":0.05090408005164609,"train/train/tensor_grad_model_layers_18_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/max_abs":0.083984375,"train/train/tensor_act_model_layers_76_self_attn_q_proj/max_abs":1.078125,"train/train/layer__model_layers_57/param/norm":17.939527868822246,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_mlp_gate_proj_weight/mean":4.3392181396484375e-05,"train/train/tensor_act_model_layers_83_mlp/mean":0.000812530517578125,"train/train/tensor_act_model_layers_37_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/mean":-5.53131103515625e-05,"train/train/tensor_act_model_layers_88_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_0/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_25/grad/mean":-1.3969548034965527e-06,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/norm":3.59375,"train/train/tensor_act_model_layers_93_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_61_input_layernorm/std":1.000006244818506,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_86/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_gate_proj/mean":0.00374603271484375,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_58_mlp_gate_proj/mean":0.0005311965942382812,"train/train/tensor_act_model_layers_80_mlp_gate_proj/max_abs":1.1015625,"train/train/tensor_grad_model_layers_23_mlp_gate_proj_weight/std":0.00014777729136572994,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn_v_proj/mean":-0.00679779052734375,"train/train/tensor_param_model_layers_21_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_23/act/std":0.41627650154086676,"train/train/tensor_param_model_layers_36_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_13/act/mean":-0.005547795976911273,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/max_abs":0.09619140625,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/max_abs":0.0703125,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/max_abs":1.341104507446289e-05,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/max_abs":0.0009613037109375,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/max_abs":0.07861328125,"train/train/tensor_act_model_layers_16_mlp/std":0.01568689766045148,"train/train/layer_model_layers_74/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/max_abs":0.0022735595703125,"train/train/tensor_act_model_layers_17_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn/norm":291.0904437653292,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/norm":0.026805901300070566,"train/train/tensor_param_model_layers_10_mlp_gate_proj_weight/max_abs":0.07958984375,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/mean":8.630752563476562e-05,"train/train/tensor_grad_model_layers_39_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/max_abs":0.0040283203125,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_59/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/mean":5.650520324707031e-05,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/norm":2.59375,"train/train/tensor_act_model_layers_32_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn_o_proj/norm":288.09591313449556,"train/train/layer_model_layers_4/grad/mean":-1.7103918881828253e-06,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_14_self_attn/norm":283.3354800938247,"train/train/tensor_act_model_layers_23_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_input_layernorm/norm":5792.433593758901,"train/train/tensor_act_model_layers_80_input_layernorm/max_abs":4.65625,"train/train/tensor_act_model_layers_21/frac_near_dtype_limit":0,"train/train/layer_model_layers_34/grad/norm":0.22854224404325144,"train/train/tensor_act_model_layers_10_mlp_down_proj/std":0.015549359624922337,"train/train/tensor_act_model_layers_69_mlp_gate_proj/mean":0.0107574462890625,"train/train/tensor_act_model_layers_16_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_90/param/mean":0.0016026965541512286,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_12_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_mlp/max_abs":0.0830078125,"train/train/tensor_act_model_layers_11_self_attn_v_proj/norm":1328.0350916201585,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/norm":0.029150981034477136,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/norm":0.0022828596752548246,"train/train/tensor_act_model_layers_48_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/max_abs":0.005096435546875,"train/train/tensor_grad_model_layers_42_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/mean":0.00016117095947265625,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/norm":0.0020013459378072345,"train/train/layer_model_layers_14/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_v_proj/std":0.21802425862161656,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/max_abs":0.08837890625,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/std":0.00018510869015835214,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/std":6.697416469926815e-07,"train/train/tensor_act_model_layers_73_self_attn_q_proj/std":0.23194101889545365,"train/train/layer_model_layers_20/grad/max_abs":0.00970458984375,"train/train/layer_model_layers_70/grad/std":0.00020089643265325385,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_88_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn_v_proj/norm":1339.1262598143544,"train/train/tensor_act_model_layers_41_self_attn/std":0.04992779007769253,"train/train/tensor_act_model_layers_36_input_layernorm/std":1.000006368712731,"train/train/tensor_grad_model_layers_65_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/norm":0.11178681760876356,"train/train/tensor_act_model_layers_26_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer__model_layers_70/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_42/grad/mean":5.371886953513913e-07,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/mean":0.0001316070556640625,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/mean":0.00013637542724609375,"train/train/tensor_act_model_layers_4_self_attn_v_proj/max_abs":1.21875,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/norm":3.65625,"train/train/tensor_act_model_layers_35_self_attn/std":0.04602314305631993,"train/train/layer__model_layers_46/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_mlp_gate_proj_weight/max_abs":0.00156402587890625,"train/train/tensor_act_model_layers_67_mlp_down_proj/std":0.015671192919477207,"train/train/tensor_param_model_layers_37_input_layernorm_weight/std":0,"train/train/layer_model_layers_39/act/mean":-0.005127889769417899,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_k_proj/std":0.23291405081939295,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_12_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82/std":0.45557157553410915,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/mean":0.00019550323486328125,"train/train/tensor_param_model_layers_9_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/mean":3.990135155618191e-08,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/max_abs":1.0669231414794922e-05,"train/train/tensor_param_model_layers_73_self_attn_k_proj_weight/max_abs":0.08447265625,"train/train/tensor_act_model_layers_5_self_attn_q_proj/norm":1270.3371515348365,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44/mean":-0.00222015380859375,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14/mean":-0.0095672607421875,"train/train/layer_model_layers_9/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_post_attention_layernorm/std":1.0000078253358065,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_k_proj/max_abs":1.140625,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/max_abs":0.09716796875,"train/train/tensor_act_model_layers_64_self_attn_q_proj/mean":0.004688262939453125,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_32_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_input_layernorm/std":1.0000001615844536,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/std":0.00039906253534278827,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/norm":0.08944814076759665,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/max_abs":0.08251953125,"train/train/tensor_act_model_layers_31_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_81_self_attn/std":0.05072163776812896,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/mean":-2.668239176273346e-06,"train/train/tensor_param_model_layers_47_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_33/mean":-0.00629425048828125,"train/train/tensor_act_model_layers_37_self_attn/norm":279.51952065436603,"train/train/tensor_grad_model_layers_64_mlp_gate_proj_weight/std":8.251423574207159e-05,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/max_abs":0.0004405975341796875,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_input_layernorm/std":1.0000070546854107,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/norm":0.24985929344165894,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/norm":0.4109715668804591,"train/train/tensor_act_model_layers_61_self_attn_o_proj/mean":0.0026397705078125,"train/train/tensor_act_model_layers_92_mlp_down_proj/mean":-0.000743865966796875,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/mean":1.921318471431732e-06,"train/train/tensor_act_model_layers_2_mlp_gate_proj/mean":0.00559234619140625,"train/train/tensor_act_model_layers_15_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_81/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_mlp_gate_proj_weight/mean":-1.0300427675247192e-06,"train/train/tensor_act_model_layers_45_mlp_gate_proj/norm":1852.6425964039424,"train/train/tensor_act_model_layers_42_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_4_mlp_down_proj/norm":92.58727412352023,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/mean":8.58306884765625e-06,"train/train/tensor_act_model_layers_11_mlp_down_proj/std":0.015778411519913788,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/max_abs":0.001220703125,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/max_abs":4.827976226806641e-06,"train/train/layer_model_layers_41/grad/mean":-5.897413140130861e-07,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_93_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_73_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_31_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/mean":1.8353603081777692e-08,"train/train/tensor_act_model_layers_87_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn_k_proj/mean":0.00676727294921875,"train/train/tensor_act_model_layers_48_self_attn/mean":0.00104522705078125,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_63/norm":2294.7931739602877,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/mean":1.9073486328125e-05,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/norm":2.59375,"train/train/tensor_param_model_layers_36_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/act/max_abs":4.90625,"train/train/layer_model_layers_13/grad/std":0.0004797433335291751,"train/train/tensor_act_model_layers_13_mlp_gate_proj/std":0.22583315666780976,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/mean":-2.68664734903723e-09,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_92/param/max_abs":1,"train/train/tensor_act_model_layers_91_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/norm":9.202308582518234e-05,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_49_mlp/max_abs":0.08203125,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/std":1.680899246102441e-06,"train/train/tensor_act_model_layers_42_self_attn_k_proj/std":0.23096193667277196,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/norm":0.005568684563428924,"train/train/tensor_act_model_layers_51/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_43/param/mean":0.001523016395509336,"train/train/tensor_act_model_layers_90_self_attn_k_proj/std":0.22656454453285735,"train/train/tensor_act_model_layers_20_mlp_gate_proj/std":0.2282744422761757,"train/train/tensor_act_model_layers_42_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/std":0.019775390625,"train/train/tensor_grad_model_layers_58_mlp_gate_proj_weight/mean":4.3655745685100555e-09,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/mean":4.7497451305389404e-07,"train/train/layer_model_layers_93/act/mean":0.003341947283063616,"train/train/tensor_act_model_layers_71_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_q_proj/mean":0.0037994384765625,"train/train/tensor_param_model_layers_23_mlp_gate_proj_weight/norm":3.59375,"train/train/tensor_act_model_layers_13_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63/std":0.39600715817155113,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_64_self_attn_o_proj_weight/norm":0.10995222094393059,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_o_proj/norm":269.06969488437073,"train/train/tensor_act_model_layers_39_self_attn_v_proj/max_abs":1.09375,"train/train/tensor_act_model_layers_11_self_attn/std":0.04956111740084291,"train/train/tensor_param_model_layers_86_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_93_mlp_down_proj/norm":91.69658652261978,"train/train/tensor_act_model_layers_34_input_layernorm/std":1.0000018901173788,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/max_abs":0.000484466552734375,"train/train/tensor_act_model_layers_52_self_attn/norm":286.35864508541863,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/max_abs":0.0908203125,"train/train/tensor_act_model_layers_22_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_mlp_gate_proj_weight/norm":0.02683014108839698,"train/train/tensor_param_model_layers_89_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_87_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_down_proj/mean":0.00015376508235931396,"train/train/tensor_act_model_layers_33_mlp_down_proj/norm":86.85686982149316,"train/train/tensor_grad_model_layers_10_mlp_gate_proj_weight/mean":4.0332088246941566e-07,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/std":4.4739063796440304e-07,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_26_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/norm":0.0006591262619126114,"train/train/tensor_act_model_layers_57_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_v_proj/std":0.22949483015256145,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/std":0.00018540761638067914,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_input_layernorm/max_abs":4.6875,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_74_post_attention_layernorm/mean":0.0015892982482910156,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/norm":0.00016979286735357632,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/mean":8.242204785346985e-08,"train/train/layer__model_layers_68/param/max_abs":1,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/max_abs":1.341104507446289e-05,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/mean":2.5538611225783825e-09,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/max_abs":0.00109100341796875,"train/train/tensor_grad_model_layers_70_mlp_gate_proj_weight/mean":7.334165275096893e-09,"train/train/tensor_act_model_layers_72_input_layernorm/std":1.0000012856372371,"train/train/tensor_act_model_layers_10_self_attn_v_proj/max_abs":1.1015625,"train/train/tensor_act_model_layers_71_self_attn/mean":0.0004916191101074219,"train/train/layer_model_layers_1/grad/norm":1.1972552208084737,"train/train/tensor_param_model_layers_48_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/std":6.99196796897738e-05,"train/train/tensor_act_model_layers_48_mlp_up_proj/mean":-0.0010275840759277344,"train/train/tensor_act_model_layers_93/frac_near_dtype_limit":0,"train/train/layer__model_layers_73/param/max_abs":1,"train/train/tensor_act_model_layers_5_self_attn_o_proj/mean":-0.0015811920166015625,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_mlp_up_proj/mean":0.0026798248291015625,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_41/param/max_abs":1,"train/train/tensor_act_model_layers_26_self_attn/norm":287.8425469270435,"train/train/layer_model_layers_45/grad/max_abs":0.005523681640625,"train/train/tensor_act_model_layers_64/mean":0.0003032684326171875,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/mean":-6.151199340820312e-05,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/mean":-0.00024127960205078125,"train/train/tensor_act_model_layers_19_mlp_up_proj/norm":1826.78278928739,"train/train/tensor_act_model_layers_31_input_layernorm/max_abs":4.90625,"train/train/layer_model_layers_28/act/norm":9040.608588684805,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_v_proj/norm":1324.9072043434646,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/max_abs":0.00019073486328125,"train/train/layer_model_layers_80/grad/std":0.00018101522857562384,"train/train/tensor_act_model_layers_34_self_attn_k_proj/norm":1295.2072636989367,"train/train/tensor_grad_model_layers_38_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_k_proj/std":0.22827527108723525,"train/train/tensor_param_model_layers_68_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_q_proj/norm":1313.11320622465,"train/train/tensor_act_model_layers_52_self_attn_k_proj/mean":-0.0019884109497070312,"train/train/layer_model_layers_2/grad/mean":1.1048594986592737e-06,"train/train/tensor_param_model_layers_71_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_45_self_attn_o_proj/mean":-0.00024832645431160927,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/norm":0.0003589744647121841,"train/train/tensor_act_model_layers_4_mlp/std":0.015960915906523435,"train/train/tensor_act_model_layers_60_mlp/norm":88.295975655842,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_o_proj/mean":-0.00024941563606262207,"train/train/tensor_param_model_layers_64_mlp_gate_proj_weight/mean":8.487701416015625e-05,"train/train/tensor_grad_model_layers_12_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/mean":-1.889929990284145e-09,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/max_abs":3.933906555175781e-06,"train/train/tensor_act_model_layers_90_self_attn_k_proj/mean":-0.0095367431640625,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/mean":-5.498528480529785e-06,"train/train/layer_model_layers_47/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn/norm":284.3693547447153,"train/train/tensor_act_model_layers_31_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn/std":0.048340807649847314,"train/train/tensor_act_model_layers_92_self_attn_o_proj/mean":0.0022830963134765625,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/norm":0.2787500109480099,"train/train/layer__model_layers_43/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_51/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_mlp_up_proj/mean":-0.00039386749267578125,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/mean":-0.00020313262939453125,"train/train/layer__model_layers_8/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn_v_proj/max_abs":1.125,"train/train/tensor_act_model_layers_77_mlp/std":0.015961466057052225,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/mean":-1.1801719665527344e-05,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_v_proj/std":0.2236328471175421,"train/train/tensor_act_model_layers_47_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_up_proj/norm":1870.879374781787,"train/train/tensor_grad_model_layers_7_mlp_gate_proj_weight/norm":0.09412246465624617,"train/train/tensor_act_model_layers_33_self_attn_k_proj/std":0.23144850987728718,"train/train/tensor_act_model_layers_68_post_attention_layernorm/norm":5792.600463875928,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_56/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp_down_proj/std":0.01596088789954434,"train/train/tensor_act_model_layers_67_mlp_down_proj/max_abs":0.07666015625,"train/train/tensor_param_model_layers_88_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_post_attention_layernorm/std":1.0000050818360273,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn_q_proj/max_abs":1.265625,"train/train/tensor_act_model_layers_17_self_attn_o_proj/max_abs":0.23046875,"train/train/tensor_param_model_layers_50_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80/norm":2612.6953868526575,"train/train/tensor_act_model_layers_20_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp_up_proj/max_abs":1.0703125,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/norm":0.11523256462097509,"train/train/layer__model_layers_44/param/std":0.04428385213764318,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/mean":6.100162863731384e-07,"train/train/tensor_act_model_layers_48_self_attn_v_proj/norm":1304.5535805150414,"train/train/tensor_param_model_layers_57_mlp_gate_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_2_mlp_gate_proj_weight/mean":9.094364941120148e-07,"train/train/layer_model_layers_7/act/std":0.4124584609676081,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/max_abs":0.0859375,"train/train/tensor_act_model_layers_82_mlp_gate_proj/std":0.22192781153519914,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_11_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_mlp_gate_proj/std":0.22607930438560453,"train/train/tensor_act_model_layers_91_self_attn_k_proj/norm":1319.748921185092,"train/train/tensor_grad_model_layers_11_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/mean":0.00014495849609375,"train/train/tensor_act_model_layers_58_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/std":0.02001953125,"train/train/layer__model_layers_21/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_v_proj/max_abs":1.109375,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/norm":0.09947584589396966,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/max_abs":0.07861328125,"train/train/layer_model_layers_46/act/std":0.42018646123412484,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_k_proj_weight/max_abs":0.080078125,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/mean":3.781169652938843e-07,"train/train/layer__model_layers_4/param/mean":0.0015772136622769599,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/max_abs":0.003173828125,"train/train/tensor_act_model_layers_11/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/max_abs":0.0001926422119140625,"train/train/tensor_act_model_layers_46_self_attn/std":0.04785207381236826,"train/train/tensor_param_model_layers_47_mlp_gate_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_72/mean":0.0007476806640625,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_75/act/mean":-0.0012559890747070312,"train/train/tensor_act_model_layers_53_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn/std":0.047241562047212374,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_69_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer__model_layers_17/param/norm":17.944957073646066,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/max_abs":0.08154296875,"train/train/layer__model_layers_46/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/norm":2.546875,"train/train/layer_model_layers_15/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp_down_proj/norm":91.43993236184221,"train/train/tensor_act_model_layers_89_mlp_up_proj/mean":0.007080078125,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/mean":-1.191161572933197e-06,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/std":6.431277514749544e-05,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/mean":2.120486897183582e-09,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/norm":0.05350942088453048,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/mean":-2.421438694000244e-08,"train/train/tensor_act_model_layers_39_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_56_mlp/std":0.015335706115750525,"train/train/tensor_param_model_layers_12_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_gate_proj/max_abs":1.0859375,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_32_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/norm":0.00011678982221386069,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/norm":0.033514692742602346,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/max_abs":0.00185394287109375,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/std":0.00012349101300954154,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/mean":0.0004024505615234375,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/max_abs":0.0859375,"train/train/tensor_act_model_layers_62_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp_up_proj/mean":0.00472259521484375,"train/train/tensor_param_model_layers_44_mlp_gate_proj_weight/mean":4.673004150390625e-05,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/max_abs":0.0888671875,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_15_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/max_abs":0.0859375,"train/train/layer_model_layers_44/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn/mean":0.002811431884765625,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_up_proj/std":0.22021587507634174,"train/train/tensor_act_model_layers_60/mean":-0.00457763671875,"train/train/tensor_param_model_layers_10_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_75_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_48/act/std":0.42000955837292103,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/max_abs":0.005035400390625,"train/train/tensor_act_model_layers_55_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp_up_proj/norm":1887.666562560548,"train/train/tensor_param_model_layers_62_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_mlp_gate_proj_weight/mean":3.4924596548080444e-07,"train/train/tensor_act_model_layers_86_post_attention_layernorm/std":1.000003402115578,"train/train/tensor_act_model_layers_45_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/max_abs":0.07958984375,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/max_abs":0.00113677978515625,"train/train/tensor_act_model_layers_26_self_attn_v_proj/norm":1334.2437992758164,"train/train/tensor_act_model_layers_33_mlp_gate_proj/norm":1840.062985564579,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/mean":-1.133116711571347e-09,"train/train/tensor_act_model_layers_93_mlp_gate_proj/norm":1873.1594756139125,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/norm":0.03248584365644024,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/norm":0.00011858802076762545,"train/train/tensor_act_model_layers_20_post_attention_layernorm/max_abs":4.46875,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_gate_proj_weight/norm":0.1420109585576545,"train/train/tensor_act_model_layers_53_self_attn_k_proj/std":0.23413942886199113,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/norm":0.07175616516315597,"train/train/tensor_act_model_layers_39_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_mlp_gate_proj_weight/max_abs":0.09521484375,"train/train/layer_model_layers_64/act/norm":9208.932941506235,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/mean":0.0002040863037109375,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_58/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_input_layernorm/mean":-0.03997802734375,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_21_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/std":0.00023427952896808304,"train/train/tensor_act_model_layers_3/norm":378.67882750809997,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_84/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/mean":3.3468008041381836e-05,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/max_abs":0.000766754150390625,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/norm":0.026681091137625557,"train/train/tensor_act_model_layers_74_mlp_down_proj/max_abs":0.0849609375,"train/train/tensor_act_model_layers_76_mlp_down_proj/std":0.014969023365321023,"train/train/tensor_grad_model_layers_51_mlp_gate_proj_weight/norm":0.029116008680435353,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/norm":2.53125,"train/train/tensor_act_model_layers_59/max_abs":1.765625,"train/train/tensor_act_model_layers_64_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_gate_proj/mean":-0.0094146728515625,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/max_abs":0.00121307373046875,"train/train/layer_model_layers_9/grad/std":0.0005661448560185251,"train/train/tensor_act_model_layers_75_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/max_abs":0.0014495849609375,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/std":0.00018668530255265946,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/max_abs":7.927417755126953e-06,"train/train/tensor_param_model_layers_9_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_39_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_15_self_attn_q_proj/norm":1265.6877952113416,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/std":0.020263671875,"train/train/tensor_grad_model_norm_weight/mean":0.00022554397583007812,"train/train/tensor_act_model_layers_30/std":0.2690478474162554,"train/train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/max_abs":0.0035858154296875,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/std":9.24047398980464e-05,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_14/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/max_abs":0.0031585693359375,"train/train/tensor_grad_model_layers_6_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_post_attention_layernorm/std":1.0000005122272848,"train/train/tensor_act_model_layers_28_mlp_down_proj/mean":-0.00047469139099121094,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/norm":0.002685687755010649,"train/train/tensor_act_model_layers_71_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn_o_proj/std":0.047854814129641016,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_post_attention_layernorm/mean":0.0117950439453125,"train/train/tensor_grad_model_layers_24_mlp_gate_proj_weight/std":0.0001311094138707059,"train/train/tensor_act_model_layers_85_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_44/grad/norm":0.20544227917973887,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/max_abs":0.00018215179443359375,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/std":0.019775390625,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/mean":1.4508259482681751e-08,"train/train/tensor_act_model_layers_87_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_mlp_up_proj/std":0.22486103756005438,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_91/act/norm":9325.271889706419,"train/train/tensor_param_model_layers_11_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_47_self_attn/std":0.05157626820845178,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/mean":-5.793571472167969e-05,"train/train/tensor_act_model_layers_18_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_88/param/std":0.04424793332739998,"train/train/tensor_param_model_layers_68_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/max_abs":0.00083160400390625,"train/train/tensor_act_model_layers_17_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/mean":7.2479248046875e-05,"train/train/tensor_act_model_layers_1_self_attn_q_proj/max_abs":1.1875,"train/train/tensor_act_model_layers_48_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp_down_proj/std":0.014939193218323496,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/norm":0.001009497486553463,"train/train/layer__model_layers_69/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/mean":0.00011730194091796875,"train/train/tensor_act_model_layers_73_self_attn_q_proj/mean":-0.000217437744140625,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/mean":-1.261010766029358e-06,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_61_post_attention_layernorm/std":1.000007494796785,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/max_abs":0.07568359375,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/norm":0.00013308797585442412,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_25_self_attn/norm":263.95461867047135,"train/train/tensor_act_model_layers_86_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_mlp_gate_proj_weight/std":9.612339236049127e-05,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/mean":-1.4179022400639951e-09,"train/train/tensor_act_model_layers_90_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/max_abs":9.5367431640625e-06,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_60_self_attn_o_proj/mean":-0.0014600753784179688,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/mean":0.00023174285888671875,"train/train/tensor_act_model_layers_42_mlp_gate_proj/std":0.22436670730495387,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_85_self_attn_q_proj/max_abs":1.0703125,"train/train/tensor_act_model_layers_45_post_attention_layernorm/std":1.000004993761004,"train/train/tensor_act_model_layers_15_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp_up_proj/norm":1848.0306899059535,"train/train/layer_model_layers_12/grad/std":0.0004936145226292014,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69/mean":-0.0007017254829406738,"train/train/tensor_grad_model_layers_61_mlp_gate_proj_weight/max_abs":0.000766754150390625,"train/train/tensor_param_model_layers_93_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_59_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/max_abs":0.002166748046875,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_mlp_gate_proj_weight/std":8.285348700605786e-05,"train/train/tensor_param_model_layers_30_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57/mean":-0.000537872314453125,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/max_abs":1.3828277587890625e-05,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/norm":0.123291015625,"train/train/tensor_act_model_layers_10_mlp_up_proj/mean":-0.005615234375,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/max_abs":0.0888671875,"train/train/tensor_param_model_layers_71_mlp_gate_proj_weight/max_abs":0.08154296875,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/mean":7.497146725654602e-08,"train/train/tensor_act_model_layers_57_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/mean":3.3155083656311035e-07,"train/train/layer_model_layers_35/grad/mean":1.3634814937452258e-06,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/std":4.2641718885553975e-07,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/max_abs":1.4781951904296875e-05,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/max_abs":0.08154296875,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/mean":0.00020313262939453125,"train/train/tensor_act_model_layers_20_mlp_down_proj/norm":92.32689033386708,"train/train/tensor_act_model_layers_52_mlp_down_proj/max_abs":0.08154296875,"train/train/layer_model_layers_84/act/mean":-3.710814884730748e-05,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/mean":1.9514118321239948e-08,"train/train/tensor_act_model_layers_91_self_attn_q_proj/max_abs":1.125,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/norm":0.000316277859277562,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/norm":0.0009195754835129153,"train/train/layer_model_layers_63/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_o_proj/std":0.04919603072127065,"train/train/tensor_act_model_layers_82_mlp_gate_proj/max_abs":1.0859375,"train/train/tensor_act_model_layers_42_post_attention_layernorm/norm":5792.5808105473025,"train/train/tensor_grad_model_layers_1_mlp_gate_proj_weight/norm":0.23084758076497996,"train/train/tensor_act_model_layers_53_mlp/max_abs":0.08984375,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/std":3.728503614620242e-05,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/max_abs":3.844499588012695e-06,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/std":0.00026916267603565284,"train/train/layer_model_layers_75/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/max_abs":0.08056640625,"train/train/tensor_act_model_layers_72_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp_gate_proj/std":0.22680856852070072,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_18/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_60_self_attn_q_proj/max_abs":1.265625,"train/train/tensor_act_model_layers_41_mlp_gate_proj/norm":1810.6048485567433,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/max_abs":0.0048828125,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/std":5.048354030148407e-07,"train/train/layer__model_layers_31/param/std":0.04425000486476228,"train/train/tensor_act_model_layers_48_self_attn/norm":291.21136793218284,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/mean":3.5371631383895874e-06,"train/train/tensor_act_model_layers_35_self_attn_q_proj/std":0.21606656103129646,"train/train/tensor_act_model_layers_8/max_abs":0.75390625,"train/train/layer__model_layers_80/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_k_proj/mean":-0.00787353515625,"train/train/tensor_act_model_layers_36_mlp/norm":93.71499610338645,"train/train/tensor_act_model_layers_77_self_attn/max_abs":0.26171875,"train/train/layer_model_layers_88/grad/mean":7.523662539986478e-08,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/norm":0.09345166102239949,"train/train/layer__model_layers_63/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_mlp_up_proj/mean":-0.0027713775634765625,"train/train/tensor_act_model_layers_32_mlp_down_proj/norm":88.51529036531764,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/mean":-5.42961061000824e-07,"train/train/layer_model_layers_41/act/mean":-0.004414916038513184,"train/train/tensor_act_model_layers_54_self_attn_q_proj/mean":0.003658294677734375,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_66_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/norm":0.19363353058830185,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/mean":0.00011014938354492188,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/mean":1.560896635055542e-06,"train/train/tensor_act_model_layers_7_mlp_down_proj/norm":90.9582977715985,"train/train/tensor_act_model_layers_88_mlp_gate_proj/mean":-0.004703521728515625,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/norm":0.026157178595682114,"train/train/tensor_act_model_layers_52_mlp_down_proj/mean":0.0004982948303222656,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/max_abs":1.2695789337158203e-05,"train/train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/std":0.0005754481216980146,"train/train/tensor_param_model_layers_35_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_43_mlp_gate_proj_weight/std":0.02001953125,"train/train/layer__model_layers_57/param/std":0.0442625553294908,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_69_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp_gate_proj/norm":1854.8997593339245,"train/train/tensor_param_model_layers_29_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/global/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/std":3.713159663437966e-07,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_40_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/max_abs":0.002685546875,"train/train/tensor_act_model_layers_79_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/std":0.0005359653490767116,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn_v_proj/max_abs":1.0703125,"train/train/tensor_grad_model_layers_26_mlp_gate_proj_weight/mean":-7.045455276966095e-07,"train/train/tensor_act_model_layers_60_mlp_down_proj/norm":88.295975655842,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/norm":0.0019505491258110363,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/max_abs":7.12275505065918e-06,"train/train/tensor_act_model_layers_48_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_78/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/max_abs":0.00112152099609375,"train/train/tensor_act_model_layers_47_post_attention_layernorm/norm":5792.588256838195,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/mean":-1.4901161193847656e-07,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_mlp_down_proj/std":0.015869591286343365,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/mean":7.031485438346863e-07,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/norm":0.1344131060840356,"train/train/tensor_act_model_layers_83_self_attn/mean":-0.002841949462890625,"train/train/layer__model_layers_75/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/mean":0.00010204315185546875,"train/train/tensor_grad_model_layers_73_mlp_gate_proj_weight/std":8.000556189018379e-05,"train/train/layer_model_layers_32/grad/mean":7.083385604522344e-07,"train/train/tensor_act_model_layers_83_mlp_gate_proj/norm":1863.5747161595004,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/mean":-2.2798776626586914e-06,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_16_self_attn_v_proj/norm":1348.6132321480047,"train/train/tensor_grad_model_layers_87_mlp_gate_proj_weight/std":6.547317575654489e-05,"train/train/tensor_act_model_layers_85_post_attention_layernorm/max_abs":4.75,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_72_self_attn_o_proj/norm":290.40290013720914,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_mlp_down_proj/norm":91.52036738745716,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_22_post_attention_layernorm/std":1.0000062212154466,"train/train/tensor_act_model_layers_41_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/std":0.019775390625,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/mean":2.7179718017578125e-05,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/mean":6.332993507385254e-07,"train/train/tensor_act_model_layers_56_self_attn_k_proj/norm":1318.4610147765545,"train/train/tensor_act_model_layers_42_mlp_up_proj/std":0.22925359464388273,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/max_abs":0.07177734375,"train/train/tensor_act_model_layers_53_mlp_gate_proj/norm":1842.5956231210891,"train/train/layer__model_layers_23/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_post_attention_layernorm/max_abs":4.65625,"train/train/layer_model_layers_51/grad/norm":0.16971646941163607,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/norm":0.00335139771814455,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_24_mlp/norm":90.90465485676727,"train/train/tensor_act_model_layers_35_mlp_up_proj/std":0.22534426547372924,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/max_abs":0.0810546875,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/max_abs":0.006256103515625,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/std":4.5648852622054195e-07,"train/train/layer__model_layers_35/param/mean":0.0015332226448237618,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/mean":2.9685907065868378e-08,"train/train/tensor_act_model_layers_44_mlp_down_proj/max_abs":0.09130859375,"train/train/layer__model_layers_4/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/max_abs":0.000118255615234375,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/norm":0.0023019161574513694,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_post_attention_layernorm/mean":0.0092315673828125,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_16_mlp_gate_proj/max_abs":1.140625,"train/train/tensor_act_model_layers_42_mlp_gate_proj/max_abs":1.09375,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/mean":9.679794311523438e-05,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/std":0.0002696947476843486,"train/train/tensor_act_model_layers_9_self_attn/std":0.046692791426449914,"train/train/tensor_act_model_layers_6/norm":615.4789615685157,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_mlp_gate_proj_weight/mean":4.363059997558594e-05,"train/train/tensor_act_model_layers_23_self_attn_v_proj/norm":1352.1969001771317,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/max_abs":0.07861328125,"train/train/layer_model_layers_73/grad/mean":1.0611115832770949e-08,"train/train/tensor_act_model_layers_48_mlp_gate_proj/std":0.22583340445852454,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_post_attention_layernorm/std":1.0000007068735843,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/mean":5.683396011590958e-07,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/norm":0.001378811231856906,"train/train/tensor_grad_model_layers_39_mlp_gate_proj_weight/norm":0.03762657737593822,"train/train/tensor_param_model_layers_51_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_46_mlp_up_proj/max_abs":1.09375,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/std":0.00010402908957199086,"train/train/tensor_act_model_layers_19/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_v_proj/std":0.22827692678144967,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/max_abs":0.08251953125,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/norm":0.10630130651035341,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/max_abs":0.07958984375,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/max_abs":0.00165557861328125,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_72_input_layernorm/mean":8.96453857421875e-05,"train/train/tensor_act_model_layers_16_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp_down_proj/std":0.015274935047781182,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/std":1.4828026973272884e-06,"train/train/tensor_act_model_layers_18_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/max_abs":0.083984375,"train/train/tensor_act_model_layers_17_self_attn_k_proj/norm":1315.7736421640593,"train/train/tensor_act_model_layers_92_input_layernorm/mean":0.0124969482421875,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/mean":-0.0002956390380859375,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/max_abs":0.01434326171875,"train/train/tensor_act_model_layers_13_self_attn_v_proj/max_abs":1.2578125,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_k_proj/mean":-0.004119873046875,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_81/param/max_abs":1,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_11_mlp_gate_proj_weight/max_abs":0.07861328125,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/max_abs":9.357929229736328e-06,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_41_self_attn/max_abs":0.244140625,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/std":4.3339314416532306e-07,"train/train/tensor_act_model_layers_93_mlp_up_proj/norm":1860.3265467124197,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/max_abs":0.0888671875,"train/train/tensor_act_model_layers_74_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/max_abs":0.083984375,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/max_abs":0.000965118408203125,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/norm":0.040072915257268055,"train/train/tensor_act_model_layers_92_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp/std":0.015335860190988045,"train/train/layer__model_layers_6/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_93_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/std":5.044938515128324e-07,"train/train/tensor_grad_model_layers_37_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_6_mlp_gate_proj/norm":1833.142908009801,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_67_mlp/max_abs":0.07666015625,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/mean":-1.0505318641662598e-05,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/mean":-3.8562575355172157e-10,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/mean":-0.0001850128173828125,"train/train/tensor_act_model_layers_28_self_attn/max_abs":0.2470703125,"train/train/layer__model_layers_27/param/std":0.04422569995346424,"train/train/tensor_act_model_layers_21_input_layernorm/max_abs":4.5,"train/train/tensor_act_model_layers_59_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/norm":0.0001655153910047439,"train/train/tensor_act_model_layers_86_mlp_gate_proj/mean":0.001651763916015625,"train/train/tensor_act_model_layers_37_mlp_up_proj/max_abs":1.1953125,"train/train/tensor_act_model_layers_38_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/max_abs":0.001983642578125,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/max_abs":0.07421875,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/norm":2.53125,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_mlp_gate_proj_weight/std":7.155264071325201e-05,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/max_abs":0.00396728515625,"train/train/layer_model_layers_12/grad/max_abs":0.01068115234375,"train/train/tensor_act_model_layers_68_mlp_up_proj/max_abs":1.046875,"train/train/layer__model_layers_47/param/norm":17.9296857979302,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/mean":-0.0001678466796875,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/max_abs":0.08447265625,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/std":0.019775390625,"train/train/tensor_param_model_layers_23_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_47/act/mean":0.00117647009236472,"train/train/tensor_act_model_layers_8_mlp/max_abs":0.09375,"train/train/tensor_act_model_layers_43_self_attn_v_proj/norm":1265.870407625691,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/mean":6.961822509765625e-05,"train/train/tensor_act_model_layers_10_self_attn_k_proj/std":0.22265741633958847,"train/train/tensor_act_model_layers_81_self_attn_k_proj/norm":1294.7973688533625,"train/train/layer_model_layers_40/act/mean":0.0005484308515276228,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_gate_proj_weight/mean":-3.14321368932724e-07,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/mean":-2.454034984111786e-07,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/std":7.737448796459233e-05,"train/train/tensor_act_model_layers_41_mlp_down_proj/norm":90.19993339537602,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63/mean":0.0003681182861328125,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_93/act/norm":9348.587383186654,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/mean":1.0927578841801733e-09,"train/train/tensor_act_model_layers_26/max_abs":1.265625,"train/train/tensor_act_model_layers_89_mlp_down_proj/std":0.01564074454483639,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/grad/mean":2.4710196344975192e-08,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/norm":9.303336137871226e-05,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/max_abs":0.09033203125,"train/train/tensor_act_model_layers_30_self_attn_o_proj/std":0.04663232584854941,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/max_abs":0.08154296875,"train/train/tensor_act_model_layers_13_post_attention_layernorm/max_abs":4.6875,"train/train/tensor_act_model_layers_78_self_attn_q_proj/norm":1294.2279009810698,"train/train/tensor_act_model_layers_31_mlp_down_proj/std":0.01577782823847169,"train/train/tensor_act_model_layers_56_self_attn/max_abs":0.2373046875,"train/train/tensor_param_model_layers_64_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_75_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp/std":0.015777797024298185,"train/train/layer__model_layers_85/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/norm":0.025292868595587784,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_input_layernorm/mean":-0.0697021484375,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_93_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp_gate_proj/norm":1875.0932237126435,"train/train/tensor_param_model_layers_20_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_58_input_layernorm/std":1.0000093741252298,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/norm":0.0007616437399745522,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/norm":0.040866945633913625,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/std":0.0006067025945813219,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/max_abs":0.07763671875,"train/train/tensor_param_model_layers_50_self_attn_q_proj_weight/max_abs":0.078125,"train/train/global/param/norm":174.8432110134677,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/std":0.0198974609375,"train/train/layer_model_layers_44/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp/mean":0.00023669004440307617,"train/train/tensor_act_model_layers_17_self_attn_k_proj/max_abs":1.1015625,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/mean":-9.104609489440918e-06,"train/train/tensor_act_model_layers_58_mlp_down_proj/std":0.01580956405218063,"train/train/tensor_act_model_layers_75_mlp/max_abs":0.083984375,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_66_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn_k_proj/mean":0.00848388671875,"train/train/tensor_act_model_layers_11_mlp_down_proj/max_abs":0.0830078125,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/mean":-0.0003910064697265625,"train/train/tensor_act_model_layers_54_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/max_abs":0.004364013671875,"train/train/tensor_act_model_layers_81_input_layernorm/mean":0.01129150390625,"train/train/tensor_grad_model_layers_86_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_down_proj/norm":92.04833459688456,"train/train/layer_model_layers_85/grad/mean":-9.560494795557861e-08,"train/train/layer_model_layers_24/act/std":0.4159568077107596,"train/train/tensor_act_model_layers_37_self_attn_k_proj/mean":0.00737762451171875,"train/train/layer_model_layers_16/grad/mean":4.391704450612786e-09,"train/train/layer_model_layers_30/grad/frac_near_dtype_limit":0,"train/train/layer__model_layers_42/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_89_mlp/std":0.01564074454483639,"train/train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_o_proj/mean":0.0007867813110351562,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp_up_proj/mean":-0.001354217529296875,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp_up_proj/mean":0.0005140304565429688,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_60_mlp_gate_proj/norm":1822.3827503601387,"train/train/layer_model_layers_81/grad/max_abs":0.004974365234375,"train/train/tensor_act_model_layers_2_self_attn/std":0.031037133401281308,"train/train/tensor_act_model_layers_45_self_attn_k_proj/mean":-0.0016756057739257812,"train/train/tensor_act_model_layers_79_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20/std":0.22095369940816553,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn_o_proj/mean":-0.00030040740966796875,"train/train/tensor_act_model_layers_92_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/mean":1.7404556274414062e-05,"train/train/tensor_act_model_layers_79_self_attn_o_proj/norm":303.2164695260724,"train/train/tensor_act_model_layers_32_mlp_gate_proj/norm":1832.4554970838722,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_up_proj/mean":-0.001163482666015625,"train/train/tensor_act_model_layers_75_self_attn_v_proj/mean":-0.0004857778549194336,"train/train/tensor_act_model_layers_34_mlp_gate_proj/mean":0.0007486343383789062,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_post_attention_layernorm/norm":5792.598510748262,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/std":0.0198974609375,"train/train/layer_model_layers_72/act/max_abs":4.71875,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/max_abs":0.09423828125,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/norm":0.06759433441685456,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_39_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer_model_layers_73/act/norm":9254.1211148517,"train/train/tensor_act_model_layers_22_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer_model_layers_65/act/std":0.42517161965791944,"train/train/layer__model_layers_19/param/max_abs":1,"train/train/tensor_grad_model_layers_16_mlp_gate_proj_weight/mean":1.8887221813201904e-06,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_q_proj/max_abs":1.0703125,"train/train/tensor_act_model_layers_88_input_layernorm/std":1.0000059483395927,"train/train/tensor_act_model_layers_85_self_attn_v_proj/std":0.2246150486426478,"train/train/tensor_param_model_layers_1_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_78/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/std":9.205003341232049e-05,"train/train/tensor_act_model_layers_90_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_gate_proj_weight/max_abs":0.0029296875,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_mlp_down_proj/mean":0.0003733634948730469,"train/train/layer__model_layers_89/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_mlp_gate_proj_weight/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/norm":0.0013051004273867631,"train/train/tensor_act_model_layers_88_mlp_down_proj/max_abs":0.08203125,"train/train/tensor_act_model_layers_66_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_92/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_40_self_attn_q_proj_weight/max_abs":0.08154296875,"train/train/tensor_act_model_layers_15/max_abs":0.9140625,"train/train/tensor_act_model_layers_6_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/max_abs":0.08837890625,"train/train/tensor_act_model_layers_15_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/max_abs":0.08544921875,"train/train/tensor_act_model_layers_46_self_attn_q_proj/std":0.223884877315034,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41/norm":1827.31706808637,"train/train/tensor_act_model_layers_80_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp_gate_proj/std":0.22534459288377204,"train/train/layer_model_layers_52/grad/max_abs":0.006011962890625,"train/train/layer_model_layers_62/act/norm":9195.96621696762,"train/train/tensor_param_model_layers_67_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_31_self_attn_k_proj/std":0.21948717234516849,"train/train/tensor_act_model_layers_53_mlp_down_proj/norm":88.05043513674913,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_27/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn/max_abs":0.22265625,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_53_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/std":4.462502586640744e-07,"train/train/layer_model_layers_58/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/mean":-2.7120113372802734e-05,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/mean":1.2991949915885925e-07,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/mean":-8.285045623779297e-06,"train/train/layer_model_layers_31/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_self_attn/max_abs":0.2314453125,"train/train/tensor_act_model_layers_86_input_layernorm/norm":5792.589843756452,"train/train/tensor_act_model_layers_69_mlp_down_proj/mean":-0.0003352165222167969,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/max_abs":0.0010986328125,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/max_abs":0.076171875,"train/train/tensor_act_model_layers_85_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp_up_proj/max_abs":1.1484375,"train/train/tensor_act_model_layers_57_self_attn_v_proj/std":0.2246152082753465,"train/train/layer_model_layers_7/grad/norm":0.5420243107431937,"train/train/tensor_act_model_layers_74_mlp_gate_proj/norm":1861.6094628018036,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_0/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_q_proj/mean":0.01727294921875,"train/train/tensor_act_model_layers_62_mlp_up_proj/mean":0.00982666015625,"train/train/tensor_act_model_layers_38_self_attn_v_proj/norm":1325.4673428698775,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_19_self_attn/mean":-0.0011930465698242188,"train/train/tensor_act_model_layers_1_self_attn/max_abs":0.224609375,"train/train/tensor_act_model_layers_83_mlp_down_proj/norm":91.16836865058079,"train/train/tensor_act_model_layers_75_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp/norm":90.59842211152406,"train/train/tensor_act_model_layers_88_self_attn_q_proj/std":0.2226640431300743,"train/train/tensor_grad_model_layers_42_mlp_gate_proj_weight/norm":0.03388391312739377,"train/train/tensor_act_model_layers_48_self_attn_k_proj/std":0.21997164028624222,"train/train/tensor_param_model_layers_16_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_down_proj/mean":-6.842613220214844e-05,"train/train/tensor_act_model_layers_67_mlp_down_proj/mean":0.0004515647888183594,"train/train/tensor_act_model_layers_18_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/norm":3.59375,"train/train/tensor_act_model_layers_93_self_attn_q_proj/max_abs":1.1015625,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/max_abs":0.08251953125,"train/train/tensor_param_model_layers_2_mlp_gate_proj_weight/mean":4.839897155761719e-05,"train/train/tensor_act_model_layers_89_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_mlp_gate_proj_weight/mean":9.393692016601562e-05,"train/train/tensor_act_model_layers_3/mean":-0.0025730133056640625,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/global/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_41/grad/max_abs":0.006256103515625,"train/train/tensor_param_model_layers_80_mlp_gate_proj_weight/max_abs":0.0888671875,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/mean":-0.0001239776611328125,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/mean":4.1117891669273376e-07,"train/train/tensor_act_model_layers_79/norm":2583.144944586833,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/max_abs":0.095703125,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/mean":2.207234501838684e-07,"train/train/tensor_act_model_layers_51_self_attn_o_proj/mean":0.0020809173583984375,"train/train/layer__model_layers_11/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_20_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/norm":0.11480659350889415,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/mean":-3.814697265625e-05,"train/train/tensor_act_model_layers_3_self_attn_v_proj/norm":1322.434114170883,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/mean":1.493026502430439e-08,"train/train/tensor_grad_model_layers_54_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_35/param/norm":17.935090757193006,"train/train/tensor_param_model_layers_14_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/max_abs":0.07861328125,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/max_abs":0.000782012939453125,"train/train/tensor_act_model_layers_65_mlp_up_proj/norm":1868.3432173677718,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/norm":0.0003502127903330866,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/mean":-1.5854835510253906e-05,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_79/param/std":0.044238056295007634,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/norm":0.004338543598438702,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/mean":6.449408829212189e-08,"train/train/tensor_act_model_layers_42_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/max_abs":0.004852294921875,"train/train/layer_model_layers_5/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_k_proj/mean":-0.00713348388671875,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/max_abs":0.0771484375,"train/train/tensor_act_model_layers_34_mlp_down_proj/max_abs":0.0869140625,"train/train/tensor_act_model_layers_46_self_attn_v_proj/norm":1312.2909518210013,"train/train/tensor_param_model_layers_15_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/max_abs":0.003204345703125,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/std":3.1897556661210084e-05,"train/train/tensor_grad_model_layers_53_mlp_gate_proj_weight/max_abs":0.0011749267578125,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/norm":0.02788024206864366,"train/train/layer__model_layers_45/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_v_proj/std":0.23242873578779544,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_12_self_attn_v_proj/norm":1337.668412384202,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/norm":0.0007641758437267922,"train/train/tensor_act_model_layers_18_input_layernorm/norm":5792.538940434949,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/mean":9.107589721679688e-05,"train/train/tensor_act_model_layers_12_input_layernorm/std":0.9980532054364426,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/std":1.1672012649492865e-05,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_92/param/mean":0.0015814389899814743,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/std":0.00014770182062353131,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/max_abs":0.00015163421630859375,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/max_abs":0.004119873046875,"train/train/layer__model_layers_18/param/mean":0.001579296570300312,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/max_abs":0.004180908203125,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/max_abs":0.083984375,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_88_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/mean":3.0067894840613008e-09,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/norm":0.11350018621421024,"train/train/tensor_param_model_layers_47_self_attn_q_proj_weight/max_abs":0.07666015625,"train/train/layer_model_layers_23/grad/max_abs":0.006072998046875,"train/train/tensor_param_model_layers_42_mlp_gate_proj_weight/max_abs":0.08740234375,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/max_abs":0.0001888275146484375,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/max_abs":4.023313522338867e-06,"train/train/layer_model_layers_75/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/max_abs":0.08642578125,"train/train/tensor_act_model_layers_20_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_mlp_gate_proj_weight/max_abs":0.0010528564453125,"train/train/tensor_act_model_layers_47_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/norm":0.0009874430636341049,"train/train/tensor_act_model_layers_88_self_attn_o_proj/std":0.04809767991862529,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/mean":7.009506225585938e-05,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/norm":0.00010131050409391948,"train/train/tensor_act_model_layers_64_mlp_up_proj/mean":0.004741668701171875,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/norm":0.028354295533163244,"train/train/tensor_act_model_layers_83_input_layernorm/max_abs":4.71875,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/mean":4.05634636990726e-09,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/max_abs":0.0016632080078125,"train/train/tensor_act_model_layers_91_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/mean":1.1650845408439636e-06,"train/train/layer_model_layers_79/grad/frac_near_user_limit":0,"train/train/layer_model_layers_55/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_embed_tokens_weight/max_abs":0.1259765625,"train/train/tensor_param_model_layers_5_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_46_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/norm":0.08571561431990396,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_o_proj/std":0.04785207381236826,"train/train/tensor_act_model_layers_2_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_input_layernorm/max_abs":4.59375,"train/train/layer_model_layers_89/act/norm":9313.179687624994,"train/train/tensor_act_model_layers_23_self_attn/std":0.04883140046339214,"train/train/tensor_param_model_layers_79_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_82/param/max_abs":1,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_92_mlp_gate_proj/mean":0.001384735107421875,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/mean":-1.987442374229431e-06,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/max_abs":1.245737075805664e-05,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/max_abs":0.00141143798828125,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/mean":-6.723403930664062e-05,"train/train/layer_model_layers_67/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_mlp_up_proj/max_abs":1.1484375,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61/mean":-0.0019412040710449219,"train/train/tensor_act_model_layers_19_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_input_layernorm/norm":5792.604614259278,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/mean":-5.9604644775390625e-05,"train/train/tensor_act_model_layers_20/norm":1281.8748283244734,"train/train/tensor_param_model_layers_79_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_self_attn/mean":-0.00044392049312591553,"train/train/tensor_act_model_layers_11_self_attn_k_proj/max_abs":1.1015625,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_1_mlp_gate_proj_weight/std":0.0006374322437190291,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/max_abs":0.09033203125,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_input_layernorm/norm":5792.599121094954,"train/train/tensor_act_model_layers_62_self_attn_v_proj/norm":1362.8249685760877,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_11/act/norm":8948.664170295999,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/norm":0.16433402707544773,"train/train/tensor_act_model_layers_33_self_attn_q_proj/max_abs":1.0859375,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/std":0.0002486215071376893,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/mean":8.940696716308594e-06,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/max_abs":3.904104232788086e-06,"train/train/tensor_param_model_layers_10_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_62/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_gate_proj/std":0.2309587142874948,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/mean":4.723668098449707e-06,"train/train/tensor_grad_model_layers_19_post_attention_layernorm_weight/std":7.0663667839581e-05,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/std":0.00038857759517073315,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/mean":1.0542571544647217e-05,"train/train/tensor_act_model_layers_86_self_attn_q_proj/mean":0.013641357421875,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/mean":-1.9550323486328125e-05,"train/train/tensor_act_model_layers_73_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_22/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/norm":0.0483708137948327,"train/train/tensor_act_model_layers_73_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_mlp_gate_proj_weight/max_abs":0.08154296875,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_16_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_gate_proj/norm":1845.006745564406,"train/train/layer__model_layers_63/param/max_abs":1,"train/train/tensor_act_model_layers_63_self_attn_v_proj/norm":1321.5778031502048,"train/train/tensor_act_model_layers_14_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_mlp_up_proj/std":0.22729663151672722,"train/train/tensor_grad_model_layers_42_mlp_gate_proj_weight/mean":1.1315569281578064e-07,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/max_abs":0.001068115234375,"train/train/tensor_act_model_layers_8_input_layernorm/mean":-0.0849609375,"train/train/tensor_param_model_layers_67_mlp_gate_proj_weight/max_abs":0.09326171875,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn/max_abs":0.2275390625,"train/train/tensor_act_model_layers_65_post_attention_layernorm/norm":5792.59313965158,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/max_abs":0.01104736328125,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_35_post_attention_layernorm/mean":-0.0174407958984375,"train/train/tensor_grad_model_layers_54_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/mean":2.409517765045166e-05,"train/train/tensor_act_model_layers_3_self_attn_q_proj/mean":-0.00589752197265625,"train/train/tensor_act_model_layers_11/max_abs":0.8203125,"train/train/tensor_grad_model_embed_tokens_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_self_attn_o_proj/max_abs":0.20703125,"train/train/tensor_act_model_layers_12_self_attn/mean":0.0014553070068359375,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_16/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_87_mlp_down_proj/max_abs":0.08056640625,"train/train/tensor_act_model_layers_79_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_1_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_1_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_70_self_attn_k_proj/std":0.22901098381093166,"train/train/tensor_act_model_layers_71_mlp_down_proj/max_abs":0.08935546875,"train/train/layer_model_layers_27/act/mean":-0.006168978554861886,"train/train/tensor_param_model_layers_39_input_layernorm_weight/mean":1,"train/train/layer_model_layers_33/act/std":0.4175780185969026,"train/train/tensor_param_model_layers_45_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_22_self_attn/norm":272.33041138620234,"train/train/tensor_act_model_layers_80_input_layernorm/std":1.0000016156905667,"train/train/tensor_grad_model_layers_31_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/std":4.59119093416351e-07,"train/train/tensor_param_model_layers_63_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_36_self_attn_k_proj/mean":0.0023508071899414062,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_83_self_attn_k_proj/max_abs":1.1328125,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn_o_proj/max_abs":0.212890625,"train/train/tensor_act_model_layers_6_self_attn_v_proj/mean":0.00762176513671875,"train/train/tensor_act_model_layers_30_self_attn_q_proj/std":0.22413236150409202,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/max_abs":0.004608154296875,"train/train/tensor_param_model_layers_2_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn_k_proj/max_abs":1.140625,"train/train/tensor_act_model_layers_55_self_attn_v_proj/mean":0.00363922119140625,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_25/act/norm":8999.53509792912,"train/train/tensor_param_model_layers_56_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/mean":1.3932585716247559e-05,"train/train/tensor_act_model_layers_34_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_29_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/std":0.0001990772424969933,"train/train/layer__model_layers_71/param/std":0.044249158245768146,"train/train/layer__model_layers_88/param/mean":0.001538245428742931,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_input_layernorm_weight/std":0,"train/train/layer__model_layers_45/param/norm":17.927888322038935,"train/train/tensor_grad_model_layers_16_mlp_gate_proj_weight/norm":0.06383110216871217,"train/train/tensor_grad_model_layers_15_mlp_gate_proj_weight/max_abs":0.0021514892578125,"train/train/tensor_act_model_layers_51_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/mean":1.9310973584651947e-06,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/max_abs":0.0022430419921875,"train/train/tensor_grad_model_layers_19_mlp_gate_proj_weight/std":0.0001602241490415273,"train/train/tensor_act_model_layers_56_self_attn_q_proj/norm":1307.7501384150273,"train/train/tensor_act_model_layers_87_mlp_up_proj/mean":-0.00650787353515625,"train/train/tensor_act_model_norm/max_abs":4.375,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/std":3.42039307199125e-07,"train/train/tensor_act_model_layers_75_input_layernorm/max_abs":4.6875,"train/train/tensor_act_model_layers_48_mlp_gate_proj/max_abs":1.125,"train/train/layer__model_layers_31/param/mean":0.0016315597081891088,"train/train/tensor_act_model_layers_10_mlp_up_proj/max_abs":1.15625,"train/train/tensor_param_model_layers_11_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_32_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/max_abs":0.09033203125,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_30_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/max_abs":0.000675201416015625,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_35_self_attn_v_proj/mean":-0.013427734375,"train/train/tensor_act_model_layers_57_self_attn_q_proj/mean":-0.009307861328125,"train/train/layer_model_layers_47/act/std":0.4214705940844176,"train/train/tensor_act_model_layers_10_mlp_gate_proj/mean":0.0077972412109375,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/max_abs":0.0927734375,"train/train/tensor_act_model_layers_14_self_attn_o_proj/mean":-0.0006768703460693359,"train/train/tensor_act_model_layers_50_mlp_gate_proj/norm":1872.813177885406,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/max_abs":0.00081634521484375,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/mean":3.180466592311859e-07,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/norm":2.5625,"train/train/layer_model_layers_16/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_down_proj/norm":90.57736296760045,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/norm":0.11116363687880824,"train/train/layer__model_layers_17/param/max_abs":1,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/mean":-3.1739473342895508e-06,"train/train/tensor_act_model_layers_71_self_attn_k_proj/mean":1.2636184692382812e-05,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/mean":0.00015735626220703125,"train/train/tensor_act_model_layers_42/max_abs":1.609375,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/mean":-3.369990736246109e-06,"train/train/layer_model_layers_93/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_75_mlp_gate_proj/std":0.22900719668234773,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_q_proj/max_abs":1.125,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/max_abs":9.834766387939453e-06,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_up_proj/norm":1841.3820408070587,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_self_attn/norm":275.98644891460725,"train/train/layer_model_layers_23/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/std":7.956151756336214e-05,"train/train/tensor_act_model_layers_55/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/max_abs":0.0038909912109375,"train/train/tensor_param_model_layers_72_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_84_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/mean":-4.674075171351433e-08,"train/train/tensor_act_model_layers_89_self_attn_v_proj/mean":-0.003154754638671875,"train/train/layer_model_layers_0/grad/max_abs":0.030029296875,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/std":0.0003881996979605865,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/mean":-1.4204852050170302e-07,"train/train/layer_model_layers_60/act/max_abs":4.625,"train/train/layer__model_layers_77/param/max_abs":1,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_83/norm":2652.7851759994796,"train/train/tensor_act_model_layers_17_self_attn_q_proj/norm":1307.0173505706466,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_5_self_attn/norm":262.9886154951609,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/std":9.331957055980674e-05,"train/train/tensor_param_model_layers_22_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_89_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_input_layernorm/mean":0.00897979736328125,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/std":0.000506706089097083,"train/train/tensor_act_model_layers_52/mean":-0.0010814666748046875,"train/train/tensor_act_model_layers_83_mlp/max_abs":0.08056640625,"train/train/tensor_act_model_layers_17_mlp_down_proj/max_abs":0.09033203125,"train/train/tensor_act_model_layers_78_self_attn_k_proj/norm":1288.0027720110081,"train/train/tensor_act_model_layers_69/norm":2415.2016149835913,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/std":0.00043276141433486886,"train/train/tensor_grad_model_layers_39_mlp_gate_proj_weight/mean":-4.1711609810590744e-07,"train/train/tensor_act_model_layers_81_post_attention_layernorm/mean":0.0093994140625,"train/train/tensor_grad_model_layers_75_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_mlp_gate_proj_weight/max_abs":0.09228515625,"train/train/tensor_param_model_layers_73_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/mean":-1.7291768017457798e-10,"train/train/tensor_act_model_layers_4_mlp_down_proj/std":0.015960915906523435,"train/train/tensor_act_model_layers_60_mlp_gate_proj/std":0.2224168924409215,"train/train/tensor_param_model_layers_43_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_self_attn_o_proj/mean":-0.0011615753173828125,"train/train/layer_model_layers_81/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/norm":9.187626298903319e-05,"train/train/tensor_act_model_layers_47_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/std":6.879647602548256e-05,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/std":0.00011710207742038887,"train/train/tensor_grad_model_layers_64_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_k_proj_weight/max_abs":0.07568359375,"train/train/tensor_act_model_layers_55_post_attention_layernorm/std":1.0000077482533185,"train/train/tensor_act_model_layers_43_mlp_gate_proj/max_abs":1.1796875,"train/train/layer__model_layers_24/param/mean":0.0015234389282797875,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/std":0.0005567278684040937,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/max_abs":0.08203125,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/mean":-1.6402918845415115e-07,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/mean":-4.2514875531196594e-07,"train/train/tensor_act_model_layers_77_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_23/grad/std":0.00037419468821583193,"train/train/tensor_param_model_layers_12_mlp_gate_proj_weight/mean":-2.7418136596679688e-05,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/norm":0.11330045339993615,"train/train/tensor_act_model_layers_81_self_attn_q_proj/norm":1311.9063509979007,"train/train/layer_model_layers_80/grad/norm":0.1466553599432951,"train/train/tensor_act_model_layers_60/max_abs":1.7890625,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/mean":0.00012969970703125,"train/train/tensor_param_model_layers_45_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_v_proj/mean":-0.002593994140625,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_k_proj/max_abs":1.1875,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/mean":-7.939888746477664e-10,"train/train/tensor_act_model_layers_7_post_attention_layernorm/max_abs":5,"train/train/tensor_act_model_layers_82_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/norm":0.00016157041514866798,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_down_proj_weight/mean":4.744529724121094e-05,"train/train/layer__model_layers_25/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_31_mlp_down_proj_weight/norm":0.04241341747187648,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/mean":4.4155967771075666e-10,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn/mean":0.00122833251953125,"train/train/tensor_grad_model_layers_79_self_attn_k_proj_weight/norm":9.505638418999704e-05,"train/train/tensor_act_model_layers_10_mlp/mean":0.00015376508235931396,"train/train/tensor_grad_model_layers_63_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_post_attention_layernorm/std":1.000005808023575,"train/train/tensor_act_model_layers_36_mlp_up_proj/max_abs":1.203125,"train/train/tensor_act_model_layers_12_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_embed_tokens_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_43/grad/max_abs":0.004486083984375,"train/train/tensor_act_model_layers_6_input_layernorm/std":0.9961063384214317,"train/train/tensor_param_model_layers_28_input_layernorm_weight/std":0,"train/train/layer__model_layers_19/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_self_attn_o_proj/norm":278.5058196344276,"train/train/layer_model_layers_7/grad/max_abs":0.015625,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/max_abs":0.0859375,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/norm":0.02936323914569913,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/std":8.346898716231319e-05,"train/train/tensor_act_model_layers_61_self_attn_k_proj/std":0.2251024870188641,"train/train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_68/param/norm":17.942705309627225,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/norm":0.00011407899250557787,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/mean":-3.128661774098873e-10,"train/train/tensor_act_model_layers_91_mlp_up_proj/mean":0.0011324882507324219,"train/train/layer_model_layers_8/grad/norm":0.49870545388952725,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_74_self_attn_v_proj/norm":1334.2364387445962,"train/train/tensor_grad_model_layers_20_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_input_layernorm/mean":-0.0543212890625,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/mean":-4.267692565917969e-05,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/std":5.660940002358781e-07,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/mean":0.00022029876708984375,"train/train/tensor_act_model_layers_79_mlp_down_proj/norm":89.57978686088487,"train/train/tensor_grad_model_norm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/norm":0.00206641133295166,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/mean":-6.020069122314453e-06,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_85/act/norm":9310.43059549836,"train/train/layer__model_layers_70/param/max_abs":1,"train/train/tensor_param_model_layers_5_mlp_gate_proj_weight/max_abs":0.08642578125,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/max_abs":0.08935546875,"train/train/tensor_act_model_layers_62_input_layernorm/max_abs":4.625,"train/train/tensor_param_model_layers_18_mlp_gate_proj_weight/std":0.0201416015625,"train/train/layer_model_layers_49/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86/std":0.4702256809007794,"train/train/layer_model_layers_3/act/norm":8911.675050416228,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/mean":-8.732080459594727e-06,"train/train/tensor_act_model_layers_37_input_layernorm/max_abs":5,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_mlp_up_proj/mean":-0.0138397216796875,"train/train/tensor_act_model_layers_51/std":0.3535264703258123,"train/train/tensor_act_model_layers_66_self_attn_o_proj/max_abs":0.2412109375,"train/train/tensor_grad_model_layers_15_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/norm":0.2237964015136176,"train/train/tensor_act_model_layers_13_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/mean":-4.380126483738422e-09,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/norm":0.00034214320475481515,"train/train/layer__model_layers_5/param/mean":0.00151050109387188,"train/train/tensor_act_model_layers_76_mlp_gate_proj/max_abs":1.046875,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/mean":-1.8826540326699615e-09,"train/train/tensor_act_model_layers_35_self_attn_o_proj/std":0.04602314305631993,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/mean":5.0067901611328125e-05,"train/train/tensor_act_model_layers_55_mlp_up_proj/max_abs":1.234375,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/std":0.0001010100440248095,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/norm":0.31505758818748,"train/train/tensor_act_model_layers_24/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/mean":-6.0498714447021484e-06,"train/train/tensor_act_model_layers_23/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_gate_proj_weight/max_abs":0.0859375,"train/train/tensor_act_model_layers_5_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_1/act/mean":-0.006672058786664691,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/mean":9.894371032714844e-06,"train/train/tensor_grad_model_layers_34_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/norm":0.00017690208717240677,"train/train/tensor_act_model_layers_93/mean":0.0091705322265625,"train/train/tensor_act_model_layers_8_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_mlp_down_proj/max_abs":0.083984375,"train/train/tensor_param_model_layers_36_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_15_self_attn_q_proj/std":0.21826318019232585,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/max_abs":0.0791015625,"train/train/tensor_act_model_layers_60_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_9_input_layernorm/max_abs":5,"train/train/tensor_act_model_layers_39_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/std":0.00032814610161943295,"train/train/tensor_param_model_layers_57_mlp_gate_proj_weight/mean":8.64267349243164e-06,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/mean":-2.3096799850463867e-07,"train/train/tensor_act_model_layers_31_post_attention_layernorm/norm":5792.577880867581,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_64/grad/norm":0.15777612726834853,"train/train/tensor_param_model_layers_29_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_81_input_layernorm/norm":5792.589111330373,"train/train/tensor_act_model_layers_81_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_86/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/max_abs":0.00011682510375976562,"train/train/tensor_act_model_layers_51_self_attn_q_proj/max_abs":1.078125,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/std":3.425117235547019e-05,"train/train/layer_model_layers_81/grad/mean":-9.537652147971616e-08,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn_v_proj/max_abs":1.1875,"train/train/tensor_act_model_layers_9_mlp_gate_proj/std":0.22656383904521782,"train/train/tensor_grad_model_layers_26_mlp_gate_proj_weight/norm":0.04258294734419081,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/std":8.34574707177868e-05,"train/train/tensor_param_model_layers_20_mlp_gate_proj_weight/mean":-0.0002651214599609375,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/max_abs":1.4781951904296875e-05,"train/train/tensor_act_model_layers_6_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/std":0.00019053648564531297,"train/train/layer_model_layers_61/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_67/max_abs":1.9375,"train/train/tensor_act_model_layers_77_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_k_proj/std":0.23120807760251588,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_mlp/max_abs":0.0791015625,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp/norm":89.93132066212658,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/std":5.04595322050463e-07,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/std":0.00040008219027924414,"train/train/layer__model_layers_73/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/mean":1.3716518878936768e-05,"train/train/tensor_param_model_layers_20_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/max_abs":0.08349609375,"train/train/tensor_act_model_layers_24_self_attn_o_proj/mean":0.0004239082336425781,"train/train/tensor_act_model_layers_55_self_attn/norm":284.8669635533363,"train/train/layer_model_layers_63/act/mean":-0.0003237128257751465,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_v_proj/norm":1237.9427446035372,"train/train/tensor_param_model_layers_79_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/max_abs":0.091796875,"train/train/tensor_param_model_layers_84_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_49_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_37/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp_gate_proj/mean":-0.01275634765625,"train/train/tensor_param_model_layers_1_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/mean":6.679445505142212e-06,"train/train/tensor_act_model_layers_92_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/max_abs":0.004913330078125,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/std":0.00012094689019013664,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/mean":-7.449125405400991e-07,"train/train/tensor_param_model_layers_66_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_down_proj_weight/mean":1.633167266845703e-05,"train/train/layer__model_layers_55/param/std":0.044246620128287076,"train/train/tensor_act_model_layers_40_self_attn_k_proj/mean":-0.0014543533325195312,"train/train/tensor_param_model_layers_42_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_post_attention_layernorm/mean":0.0031120777130126953,"train/train/layer_model_layers_50/act/mean":0.005055797951562064,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_89_mlp_gate_proj/norm":1840.780128038496,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/norm":0.002612908908450966,"train/train/tensor_act_model_layers_41_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48/mean":-6.866455078125e-05,"train/train/layer__model_layers_42/param/mean":0.0015140182328484545,"train/train/tensor_act_model_layers_44_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_24_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_mlp_down_proj/norm":92.51323163452432,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/max_abs":0.001068115234375,"train/train/tensor_act_model_layers_71_mlp_up_proj/mean":0.0009722709655761719,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/norm":0.04073437126578141,"train/train/tensor_act_model_layers_12_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18/mean":-0.01092529296875,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn_q_proj/mean":0.0015103816986083984,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/std":0.00011533972391328564,"train/train/tensor_act_model_layers_72_mlp_up_proj/std":0.22851820823542035,"train/train/tensor_act_model_layers_48_self_attn_o_proj/max_abs":0.220703125,"train/train/tensor_param_model_layers_37_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn_v_proj/std":0.2290121018055504,"train/train/tensor_act_model_layers_85_mlp_up_proj/std":0.2280281853609258,"train/train/tensor_param_model_layers_19_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp/std":0.015274157458811454,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/mean":1.409091055393219e-06,"train/train/tensor_act_model_layers_53_mlp_down_proj/std":0.015198404618691739,"train/train/tensor_act_model_layers_31_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/norm":0.23154263924744586,"train/train/tensor_grad_model_layers_90_self_attn_k_proj_weight/max_abs":4.559755325317383e-06,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_93/grad/norm":0.14944015539492292,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_69/param/std":0.04425361396602504,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_20_self_attn/norm":287.3581508952125,"train/train/tensor_act_model_layers_89_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/mean":-0.00015163421630859375,"train/train/tensor_act_model_layers_28_self_attn/std":0.05066255488261313,"train/train/tensor_act_model_layers_44_self_attn/max_abs":0.2275390625,"train/train/tensor_grad_model_layers_51_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_26/act/std":0.4164178445069281,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/std":8.449337170397458e-07,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/mean":-0.00014495849609375,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/std":0.000136089130452205,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/std":6.870181502209463e-05,"train/train/layer_model_layers_51/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22/frac_near_dtype_limit":0,"train/train/layer_model_layers_84/grad/max_abs":0.0031890869140625,"train/train/layer_model_layers_51/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/max_abs":0.0869140625,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/norm":0.029967142927181504,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/norm":0.14910897416164162,"train/train/tensor_act_model_layers_80_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/std":4.5425350439281925e-07,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/mean":2.905726432800293e-06,"train/train/layer__model_layers_2/param/mean":0.0015509214118564743,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/norm":0.026096009139528024,"train/train/tensor_act_model_layers_62_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/max_abs":0.08837890625,"train/train/tensor_act_model_layers_4_mlp_up_proj/norm":1880.7785426247883,"train/train/tensor_act_model_layers_58_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_mlp/std":0.015808779458642157,"train/train/tensor_act_model_layers_69_mlp_up_proj/mean":0.0101776123046875,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp/norm":91.92082456292871,"train/train/tensor_param_model_layers_56_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_77/act/std":0.42761818139913194,"train/train/tensor_act_model_layers_72_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_self_attn_o_proj/mean":0.0006890296936035156,"train/train/tensor_act_model_layers_42_mlp/norm":89.32513311640379,"train/train/tensor_act_model_layers_90_self_attn_o_proj/norm":301.4854531791899,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/max_abs":0.087890625,"train/train/tensor_act_model_layers_52_self_attn_o_proj/norm":286.35864508541863,"train/train/tensor_act_model_layers_70_input_layernorm/std":1.000001521290167,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_25_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp_up_proj/mean":0.00487518310546875,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/max_abs":0.00118255615234375,"train/train/tensor_act_model_layers_32_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_v_proj/mean":0.002063751220703125,"train/train/tensor_param_model_layers_89_mlp_gate_proj_weight/max_abs":0.0849609375,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/mean":0.000293731689453125,"train/train/tensor_act_model_layers_13_mlp/mean":-7.62939453125e-06,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/std":2.3992240408956804e-06,"train/train/tensor_param_model_layers_60_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/std":0.0006758841530299308,"train/train/tensor_act_model_layers_27/max_abs":1.2734375,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_down_proj/norm":89.78689562465455,"train/train/layer_model_layers_32/act/max_abs":5.03125,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/norm":0.0001060882308144767,"train/train/layer__model_layers_92/param/std":0.044246376012268815,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_55_self_attn_k_proj/std":0.22706112786785915,"train/train/layer_model_layers_45/grad/norm":0.19354740456672492,"train/train/tensor_act_model_layers_77_self_attn_v_proj/std":0.22901140064891662,"train/train/tensor_grad_model_layers_92_mlp_gate_proj_weight/max_abs":0.00102996826171875,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/max_abs":0.000789642333984375,"train/train/tensor_grad_model_layers_47_mlp_gate_proj_weight/mean":-3.1921081244945526e-07,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/mean":-2.723027137108147e-09,"train/train/tensor_grad_model_layers_52_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/mean":8.87364149093628e-06,"train/train/tensor_grad_model_layers_67_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_83_self_attn/norm":286.67480820701564,"train/train/tensor_act_model_layers_25_self_attn_k_proj/max_abs":1.0625,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/std":6.190128309119637e-05,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/std":0.00014680795105032375,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/mean":7.627531886100769e-07,"train/train/layer_model_layers_22/grad/max_abs":0.00799560546875,"train/train/tensor_act_model_layers_53_mlp_up_proj/max_abs":1.0546875,"train/train/tensor_act_model_layers_56_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_post_attention_layernorm/mean":-0.011474609375,"train/train/tensor_act_model_layers_85_mlp/std":0.01596088789954434,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/max_abs":0.001190185546875,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_3_mlp_gate_proj_weight/mean":1.6130506992340088e-06,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/max_abs":0.08349609375,"train/train/tensor_act_model_layers_73/std":0.4292074019216187,"train/train/tensor_act_model_layers_32_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_input_layernorm/mean":0.0037565231323242188,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_up_proj/std":0.22437222584550046,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_post_attention_layernorm/max_abs":4.8125,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/mean":2.2910535335540771e-07,"train/train/tensor_act_model_layers_86_mlp/std":0.01568770068375377,"train/train/tensor_act_model_layers_37_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_24_mlp_down_proj/std":0.01570165041606858,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_v_proj/mean":-0.0019025802612304688,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn/std":0.04388577182808841,"train/train/tensor_param_model_layers_66_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/mean":-1.5441328287124634e-06,"train/train/tensor_act_model_layers_14_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp_gate_proj/max_abs":1.1328125,"train/train/layer_model_layers_32/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_k_proj/mean":-0.00726318359375,"train/train/tensor_act_model_layers_57_self_attn_q_proj/max_abs":1.0859375,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/std":0.0003541436373162968,"train/train/tensor_act_model_layers_40_mlp_down_proj/norm":88.92811747843332,"train/train/tensor_act_model_layers_69_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_mlp_down_proj/mean":-0.0002856254577636719,"train/train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_93/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/mean":-1.271069049835205e-05,"train/train/tensor_act_model_layers_88_mlp/norm":89.73871702312833,"train/train/tensor_param_model_layers_81_mlp_gate_proj_weight/mean":-0.00010919570922851562,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/norm":0.00011261738872193779,"train/train/tensor_act_model_layers_19_self_attn_q_proj/max_abs":1.1015625,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/mean":-1.4811121218372136e-09,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_88/grad/norm":0.1423975966958521,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/max_abs":0.00191497802734375,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/std":9.676495950007001e-05,"train/train/tensor_act_model_layers_20_mlp_up_proj/max_abs":1.140625,"train/train/tensor_act_model_layers_83_mlp_gate_proj/max_abs":1.125,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/mean":4.5183696784079075e-09,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/norm":9.535998076896704e-05,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/max_abs":0.10009765625,"train/train/layer_model_layers_29/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17/mean":-0.010955810546875,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_q_proj/std":0.22364145210079794,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/mean":6.309710443019867e-08,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/std":5.668043865677617e-07,"train/train/tensor_act_model_layers_31_self_attn_k_proj/norm":1271.792871237993,"train/train/tensor_act_model_layers_34_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn/mean":-0.0022430419921875,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_gate_proj/max_abs":1.1875,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_10_input_layernorm/norm":5792.462890631555,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92/std":0.4834048137701835,"train/train/tensor_act_model_layers_77_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_gate_proj_weight/norm":3.640625,"train/train/layer__model_layers_80/param/norm":17.928773465263873,"train/train/tensor_param_model_layers_16_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_51_self_attn_o_proj/norm":286.85756736947855,"train/train/tensor_act_model_layers_47_self_attn_q_proj/mean":0.004604339599609375,"train/train/tensor_act_model_layers_24/std":0.2409711965876197,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/max_abs":0.0859375,"train/train/tensor_act_model_layers_75/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_91_mlp_gate_proj_weight/norm":0.024150725168393734,"train/train/tensor_act_model_layers_73_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_mlp_gate_proj/std":0.2236350542025362,"train/train/tensor_act_model_layers_82_self_attn_k_proj/norm":1330.1907509217258,"train/train/layer_model_layers_3/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_gate_proj_weight/max_abs":0.00099945068359375,"train/train/tensor_act_model_layers_89_self_attn_k_proj/std":0.22534708625696193,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/mean":-3.5279663279652596e-07,"train/train/tensor_act_model_layers_70_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_v_proj/norm":1300.7745303612703,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/std":4.7048069086229413e-07,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/max_abs":0.08154296875,"train/train/tensor_param_model_layers_42_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/mean":-4.016328603029251e-08,"train/train/tensor_act_model_layers_9/std":0.14135993101268224,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/max_abs":0.0010223388671875,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/mean":-0.0002269744873046875,"train/train/tensor_act_model_layers_50_mlp_gate_proj/std":0.22876141301638742,"train/train/tensor_act_model_layers_53_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn_v_proj/max_abs":1.109375,"train/train/tensor_grad_model_layers_2_mlp_gate_proj_weight/norm":0.18513137070784633,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_o_proj/max_abs":0.24609375,"train/train/tensor_act_model_layers_84_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/max_abs":0.0859375,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_v_proj/norm":1272.6970291169805,"train/train/tensor_param_model_layers_6_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_53/param/max_abs":1,"train/train/tensor_grad_model_layers_7_mlp_gate_proj_weight/mean":8.470378816127777e-07,"train/train/tensor_act_model_layers_47_self_attn/max_abs":0.259765625,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/mean":5.833804607391357e-06,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/norm":0.00688670246124505,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/norm":0.002350163129746105,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/std":0.02001953125,"train/train/layer_model_layers_30/grad/max_abs":0.0059814453125,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/max_abs":0.00029754638671875,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/std":0.0003375620701073072,"train/train/layer__model_layers_39/param/std":0.044265341332181296,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/std":0.00043215136248688245,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/std":7.399996351434129e-07,"train/train/tensor_act_model_layers_36_self_attn_q_proj/std":0.23877956070927706,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/norm":0.035114475438304815,"train/train/tensor_act_model_layers_10_self_attn_k_proj/norm":1290.0654106240627,"train/train/layer_model_layers_2/grad/norm":1.1261024968409212,"train/train/layer_model_layers_48/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_gate_proj_weight/max_abs":0.08349609375,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/norm":0.0008852234944055399,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/mean":0.000270843505859375,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/max_abs":0.0888671875,"train/train/tensor_act_model_layers_43_mlp_up_proj/norm":1869.5403798747548,"train/train/tensor_act_model_layers_35_self_attn_v_proj/norm":1289.4668120976025,"train/train/tensor_act_model_layers_8_self_attn/std":0.048280399280631305,"train/train/layer__model_layers_24/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn/max_abs":0.212890625,"train/train/tensor_param_model_layers_88_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_42_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/norm":0.0024252236004795808,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp_gate_proj/norm":1893.2964625597335,"train/train/tensor_act_model_layers_25_input_layernorm/std":1.0000002589076422,"train/train/tensor_grad_model_layers_7_mlp_gate_proj_weight/max_abs":0.003387451171875,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/mean":1.3217329978942871e-05,"train/train/layer_model_layers_48/act/norm":9103.691334420742,"train/train/tensor_act_model_layers_62_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn_q_proj/frac_near_user_limit":0,"train/train/layer__model_layers_93/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_57/act/mean":0.001934062157358442,"train/train/tensor_act_model_layers_80/std":0.4506867421945346,"train/train/tensor_act_model_layers_90_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46/mean":-0.0009717941284179688,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/max_abs":0.00148773193359375,"train/train/tensor_param_model_layers_25_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_83_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_norm/norm":5792.595336918584,"train/train/tensor_grad_model_layers_68_mlp_gate_proj_weight/max_abs":0.00078582763671875,"train/train/tensor_act_model_layers_55_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_input_layernorm/norm":5792.598876955188,"train/train/tensor_act_model_layers_52_post_attention_layernorm/norm":5792.590087893521,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/norm":3.65625,"train/train/tensor_act_model_layers_30_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_v_proj/std":0.22583860226601443,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_93_self_attn_q_proj/mean":-0.00688934326171875,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/max_abs":0.00139617919921875,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/mean":-0.0002613067626953125,"train/train/tensor_act_model_layers_37_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_down_proj/max_abs":0.08837890625,"train/train/tensor_act_model_layers_93/norm":2809.037315925015,"train/train/tensor_act_model_layers_12_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/std":4.942963526072356e-07,"train/train/tensor_act_model_layers_33_self_attn_v_proj/norm":1290.169230647073,"train/train/layer__model_layers_22/param/max_abs":1,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_65_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/mean":-0.00011682510375976562,"train/train/tensor_act_model_layers_45_mlp/max_abs":0.0849609375,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/max_abs":0.005340576171875,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_36_mlp_gate_proj/mean":0.003536224365234375,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp_gate_proj/mean":-0.00921630859375,"train/train/tensor_param_model_layers_11_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_65/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/max_abs":0.0025634765625,"train/train/tensor_act_model_layers_31_input_layernorm/mean":-0.042236328125,"train/train/tensor_act_model_layers_34_self_attn_v_proj/std":0.22827443054664326,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/norm":0.0008157322266689836,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/mean":0.000125885009765625,"train/train/tensor_act_model_layers_0_mlp_gate_proj/mean":0.00251007080078125,"train/train/tensor_act_model_layers_43_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/max_abs":0.083984375,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_71_mlp_gate_proj_weight/max_abs":0.0009307861328125,"train/train/tensor_act_model_layers_63_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_mlp/mean":-0.0003447532653808594,"train/train/tensor_act_model_layers_47_input_layernorm/norm":5792.585449219954,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_mlp_gate_proj/std":0.22852163718921115,"train/train/tensor_act_model_layers_18_self_attn/norm":261.8512156003335,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/max_abs":0.00090789794921875,"train/train/tensor_act_model_layers_7_input_layernorm/norm":5792.343872074494,"train/train/tensor_act_model_layers_20_mlp_up_proj/norm":1860.9731055757732,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/max_abs":0.072265625,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/max_abs":0.0014190673828125,"train/train/tensor_act_model_layers_43_mlp_up_proj/std":0.2280344150230967,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/norm":0.03139497981760221,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/mean":-7.200241088867188e-05,"train/train/layer_model_layers_6/act/max_abs":5.1875,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_5_mlp_gate_proj/max_abs":1.3046875,"train/train/layer_model_layers_42/grad/std":0.00023719658037931935,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn_v_proj/mean":0.017822265625,"train/train/tensor_act_model_layers_7_mlp_gate_proj/max_abs":1.1875,"train/train/layer__model_layers_1/param/max_abs":1,"train/train/tensor_act_model_layers_73_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/mean":-8.498318493366241e-09,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_up_proj/mean":0.00862884521484375,"train/train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_10/act/frac_near_dtype_limit":0,"train/train/layer__model_layers_72/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_gate_proj/norm":1840.2128351510407,"train/train/tensor_act_model_layers_53_self_attn_q_proj/mean":0.0043792724609375,"train/train/tensor_act_model_layers_76_self_attn_q_proj/norm":1320.6524070220048,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn/std":0.049868072169304764,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_5_mlp_down_proj/norm":90.59024442844478,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/std":0.019775390625,"train/train/layer__model_layers_41/param/std":0.04424438473647816,"train/train/tensor_param_model_layers_91_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_69_mlp_gate_proj_weight/max_abs":0.0009918212890625,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/norm":2.515625,"train/train/tensor_param_model_layers_8_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_10_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_91/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/norm":2.53125,"train/train/tensor_act_model_layers_26_mlp_gate_proj/norm":1866.9616598376306,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/norm":0.0007031134887587494,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_mlp_up_proj/std":0.22558993262075658,"train/train/tensor_act_model_layers_34_mlp/norm":90.84844103582273,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/std":0.0012965908250726574,"train/train/tensor_act_model_layers_27_input_layernorm/mean":-0.04510498046875,"train/train/tensor_act_model_layers_11_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/std":9.450972289266544e-07,"train/train/tensor_act_model_layers_82_self_attn_v_proj/std":0.23195746386819266,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/max_abs":0.083984375,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/std":0.02001953125,"train/train/layer_model_layers_32/grad/norm":0.22094486798170943,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_post_attention_layernorm/std":1.0000055618429486,"train/train/tensor_act_model_layers_0_post_attention_layernorm/std":1.00000087171755,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_self_attn_q_proj_weight/max_abs":0.07763671875,"train/train/layer_model_layers_18/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_71/act/mean":-0.0004506111145019531,"train/train/tensor_act_model_layers_42/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_post_attention_layernorm/std":1.000005529617911,"train/train/tensor_act_model_layers_46/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_input_layernorm/norm":5792.487426759916,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp_down_proj/std":0.015686609768660917,"train/train/tensor_act_model_layers_44_input_layernorm/mean":-0.014007568359375,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_50/param/std":0.04422962242411968,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/mean":9.569339454174042e-08,"train/train/tensor_act_model_layers_70_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/std":0.0017139667946377125,"train/train/tensor_grad_model_layers_82_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/global/act/max_abs":8.32306957244873,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/max_abs":0.080078125,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/norm":2.59375,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/std":4.327158189320338e-07,"train/train/tensor_act_model_layers_40_self_attn_v_proj/std":0.21582301316686034,"train/train/tensor_act_model_layers_7_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_input_layernorm/mean":-0.0501708984375,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/mean":9.870529174804688e-05,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/std":0.0201416015625,"train/train/layer__model_layers_62/param/norm":17.937302644820235,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/std":0.0004688084892734084,"train/train/tensor_param_model_layers_71_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_36/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/mean":-8.775987225817516e-08,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/norm":0.0018218843980231077,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/norm":0.11202277583762137,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/norm":0.0012746798212643555,"train/train/tensor_grad_model_layers_44_mlp_gate_proj_weight/norm":0.03723964913458001,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/mean":3.504101186990738e-07,"train/train/layer_model_layers_22/act/std":0.4152045365203885,"train/train/tensor_param_model_layers_8_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_0_mlp_up_proj/std":0.22680705022210298,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/std":7.643679593235359e-05,"train/train/tensor_act_model_layers_22_self_attn_o_proj/max_abs":0.234375,"train/train/tensor_act_model_layers_50_self_attn_v_proj/max_abs":1.0390625,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93/frac_near_user_limit":0,"train/train/layer_model_layers_26/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/mean":-6.123445928096771e-07,"train/train/tensor_param_model_layers_81_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_86/grad/max_abs":0.0048828125,"train/train/tensor_param_model_layers_30_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_91_mlp_gate_proj_weight/std":6.666491514737648e-05,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/std":0.0004481153621970675,"train/train/tensor_act_model_layers_60_self_attn_k_proj/std":0.22387840079169177,"train/train/tensor_act_model_layers_86_post_attention_layernorm/norm":5792.602172851638,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_mlp_up_proj/max_abs":1.1484375,"train/train/tensor_act_model_layers_47_self_attn_o_proj/std":0.05157626820845178,"train/train/tensor_grad_model_layers_3_self_attn_o_proj_weight/norm":0.6137320447109456,"train/train/tensor_grad_model_layers_69_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_q_proj_weight/std":1.4220116473984656e-06,"train/train/tensor_act_model_layers_7/frac_near_user_limit":0,"train/train/layer__model_layers_35/param/max_abs":1,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/mean":-3.600120544433594e-05,"train/train/tensor_grad_model_layers_19_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77/std":0.4419002016001865,"train/train/tensor_act_model_layers_27_mlp_up_proj/std":0.22314589602263812,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp_up_proj/max_abs":1.1953125,"train/train/tensor_act_model_layers_56_input_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_36/param/mean":0.0015370432932551676,"train/train/tensor_act_model_layers_51_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/act/max_abs":4.96875,"train/train/tensor_act_model_layers_18_self_attn/std":0.04522819097907275,"train/train/layer_model_layers_41/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_mlp/mean":0.0002741813659667969,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/max_abs":0.005767822265625,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/std":9.371618280989923e-05,"train/train/tensor_act_model_layers_7_mlp_up_proj/mean":0.0005826950073242188,"train/train/tensor_act_model_layers_70_self_attn_k_proj/max_abs":1.0859375,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_83_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_mlp_gate_proj/mean":0.0017757415771484375,"train/train/tensor_act_model_layers_34_post_attention_layernorm/norm":5792.572875976926,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp_down_proj/norm":93.75454180365125,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/mean":-0.0001373291015625,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/mean":0.00023365020751953125,"train/train/tensor_act_model_layers_13_self_attn_q_proj/norm":1323.0548638067255,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/std":4.0212437111356364e-07,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/max_abs":0.07373046875,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_17/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/mean":-2.436339855194092e-06,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/mean":1.5106052160263062e-06,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_q_proj/std":0.22485551512219554,"train/train/tensor_act_model_layers_28_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_85_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn/max_abs":0.251953125,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/max_abs":0.01409912109375,"train/train/tensor_act_model_layers_56_self_attn_v_proj/max_abs":1.1015625,"train/train/tensor_act_model_layers_72_self_attn_o_proj/mean":0.0003476142883300781,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/max_abs":0.08154296875,"train/train/tensor_act_model_layers_87_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_input_layernorm_weight/std":0,"train/train/layer_model_layers_4/grad/norm":0.7243513826715494,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/max_abs":1.4841556549072266e-05,"train/train/tensor_grad_model_layers_5_mlp_down_proj_weight/mean":4.4354237616062164e-07,"train/train/tensor_act_model_layers_71_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_mlp_up_proj/norm":1875.6888982702862,"train/train/layer_model_layers_15/act/std":0.4148465058851277,"train/train/tensor_act_model_layers_22_self_attn_v_proj/mean":-0.00722503662109375,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn_k_proj/std":0.2236382654451339,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/std":0.0001469175633322054,"train/train/layer__model_layers_30/param/mean":0.0015196658891746294,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/max_abs":0.07958984375,"train/train/layer_model_layers_28/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/norm":0.003247407666227884,"train/train/tensor_act_model_layers_92_self_attn_v_proj/std":0.2287685321107243,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_89_mlp/max_abs":0.08447265625,"train/train/tensor_act_model_layers_23_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_mlp_gate_proj_weight/std":7.906632437921121e-05,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/max_abs":4.857778549194336e-06,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/max_abs":4.976987838745117e-06,"train/train/tensor_act_model_layers_36_post_attention_layernorm/mean":-0.00806427001953125,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_mlp_down_proj/std":0.01539644150568739,"train/train/tensor_act_model_layers_60/std":0.3847750823557835,"train/train/tensor_param_model_layers_5_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_33_self_attn_v_proj/std":0.22265765854741332,"train/train/tensor_param_model_layers_85_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_80_self_attn/mean":0.0032196044921875,"train/train/tensor_act_model_layers_3_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/mean":-2.9921531677246094e-05,"train/train/tensor_act_model_layers_10_post_attention_layernorm/std":0.9990269781145018,"train/train/layer__model_layers_39/param/norm":17.940868313974523,"train/train/layer__model_layers_34/param/std":0.04425406608247043,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_68_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_28_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_88/std":0.47413219572172577,"train/train/tensor_act_model_layers_80_mlp_up_proj/norm":1855.2879517854121,"train/train/tensor_act_model_layers_24_mlp_gate_proj/norm":1858.5956503016378,"train/train/tensor_act_model_layers_33_input_layernorm/max_abs":5.1875,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/mean":-1.41968484967947e-07,"train/train/tensor_act_model_layers_38_mlp_down_proj/max_abs":0.0888671875,"train/train/tensor_act_model_layers_34_self_attn/std":0.04852376368689302,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_gate_proj/norm":1880.5391877540183,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/norm":0.10967594705367983,"train/train/tensor_act_model_layers_44_self_attn_q_proj/mean":-0.0108184814453125,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/mean":1.1837109923362732e-06,"train/train/tensor_param_model_layers_77_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer__model_layers_25/param/norm":17.922120203204894,"train/train/tensor_act_model_layers_10_input_layernorm/mean":-0.0523681640625,"train/train/tensor_act_model_layers_88_self_attn_o_proj/mean":-0.0002859830856323242,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_29/grad/max_abs":0.00628662109375,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/max_abs":5.21540641784668e-06,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_34_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_40/grad/mean":-4.900362220550662e-07,"train/train/tensor_grad_model_layers_26_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16/max_abs":0.9765625,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_gate_proj/std":0.22632120411668052,"train/train/tensor_act_model_layers_25_mlp_up_proj/norm":1827.9512814717677,"train/train/layer_model_layers_40/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/norm":0.10425090562532276,"train/train/tensor_act_model_layers_68_self_attn_o_proj/std":0.05090408005164609,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp_gate_proj/max_abs":1.0390625,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_input_layernorm/std":1.0000062897745474,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/mean":5.730544216930866e-07,"train/train/tensor_param_model_layers_19_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/max_abs":0.005035400390625,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/mean":-3.9577484130859375e-05,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/std":0.019775390625,"train/train/tensor_act_model_layers_14/norm":1031.251073617929,"train/train/tensor_act_model_layers_37_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp_gate_proj/norm":1853.965849815443,"train/train/tensor_act_model_layers_90_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_mlp_gate_proj_weight/mean":3.2549723982810974e-07,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/mean":2.0558945834636688e-07,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/max_abs":0.07861328125,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/norm":0.13663763637338747,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/norm":0.025188743858766377,"train/train/layer__model_layers_70/param/norm":17.930638931728005,"train/train/tensor_grad_model_layers_14_input_layernorm_weight/norm":0.004541380672753581,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/std":4.318096467740458e-05,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/mean":1.0882104106713086e-09,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp_down_proj/norm":91.61138897854775,"train/train/tensor_act_model_layers_28_self_attn_k_proj/mean":0.003429412841796875,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/norm":0.0001534430561016761,"train/train/tensor_param_model_layers_32_mlp_gate_proj_weight/mean":0.00012302398681640625,"train/train/tensor_act_model_layers_1/mean":-0.001903533935546875,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/std":0.0007865091665071498,"train/train/layer_model_layers_4/grad/max_abs":0.0147705078125,"train/train/tensor_act_model_layers_59_self_attn_o_proj/norm":287.5581165163929,"train/train/layer__model_layers_55/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/max_abs":0.08740234375,"train/train/tensor_grad_model_layers_92_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp_up_proj/mean":-0.003704071044921875,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/norm":0.035478387607182846,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_q_proj/max_abs":1.1015625,"train/train/tensor_param_model_layers_5_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/mean":-3.022141754627228e-07,"train/train/tensor_act_model_layers_5_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_input_layernorm/std":1.0000092781053131,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/max_abs":0.0054931640625,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_56_mlp_up_proj/std":0.22070483576700478,"train/train/tensor_act_model_layers_52_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/mean":-4.235334927216172e-08,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/std":4.7366287545383514e-07,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_41/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/mean":9.632110595703125e-05,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/std":6.072718465474276e-07,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/norm":0.030458683405028546,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/std":0.0008093659215412289,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/mean":8.754432201385498e-08,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_q_proj/norm":1303.2787215699595,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_53_self_attn_v_proj/std":0.22021883762146627,"train/train/tensor_act_model_layers_37_self_attn_v_proj/max_abs":1.1484375,"train/train/tensor_act_model_layers_67_self_attn_k_proj/mean":-4.6372413635253906e-05,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/norm":0.11078686225136833,"train/train/tensor_grad_model_layers_58_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_3/grad/mean":1.675661438247901e-06,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_34/max_abs":1.59375,"train/train/tensor_act_model_layers_27_self_attn/norm":277.1543360672444,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn/norm":283.6611416763035,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/max_abs":0.00119781494140625,"train/train/tensor_act_model_layers_33_mlp_up_proj/norm":1833.0987027333854,"train/train/tensor_act_model_layers_45_mlp_down_proj/std":0.015808777883169454,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/std":2.41198023612394e-05,"train/train/tensor_act_model_layers_68_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/norm":3.59375,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/std":0.0009102413309422772,"train/train/layer__model_layers_92/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_v_proj/mean":0.004764556884765625,"train/train/tensor_act_model_layers_24_self_attn_o_proj/norm":269.83750986738465,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_mlp_gate_proj_weight/max_abs":0.000835418701171875,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_46_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_26/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_4_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_self_attn_q_proj_weight/max_abs":0.0849609375,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/max_abs":0.00095367431640625,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_q_proj/norm":1297.941992851612,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/std":0.02001953125,"train/train/layer__model_layers_26/param/norm":17.92880069974704,"train/train/tensor_grad_model_layers_10_mlp_gate_proj_weight/std":0.00020265154948529134,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/max_abs":0.000396728515625,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/max_abs":4.976987838745117e-06,"train/train/tensor_act_model_layers_81_mlp_up_proj/std":0.22412292573314888,"train/train/tensor_act_model_layers_12_mlp/mean":0.00024651363492012024,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/mean":1.1368683772161603e-09,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_k_proj/mean":0.005786895751953125,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/norm":0.0001131900835782728,"train/train/tensor_act_model_layers_9_mlp_up_proj/mean":-0.009429931640625,"train/train/tensor_act_model_layers_63_input_layernorm/norm":5792.594482425132,"train/train/tensor_act_model_layers_10_self_attn_q_proj/norm":1298.0890017257382,"train/train/tensor_param_model_layers_57_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_self_attn_k_proj/max_abs":1.1796875,"train/train/tensor_act_model_layers_13_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/max_abs":0.0810546875,"train/train/tensor_param_model_layers_17_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/std":0.02001953125,"train/train/layer_model_layers_50/act/norm":9136.690487614722,"train/train/layer__model_layers_12/param/std":0.04424486787570123,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/mean":0.00014400482177734375,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/max_abs":0.0810546875,"train/train/layer_model_layers_29/act/norm":9029.595503848695,"train/train/tensor_param_model_layers_75_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_60_mlp_gate_proj_weight/max_abs":0.07958984375,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/std":0.0003838537036314861,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/max_abs":0.0859375,"train/train/tensor_act_model_layers_7_self_attn/std":0.04663394474949239,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/norm":0.0007707340705423482,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/norm":0.00290123715232879,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/mean":0.00016021728515625,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/mean":2.955552190542221e-06,"train/train/tensor_act_model_layers_67_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_49/act/mean":0.0018849372863769531,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_23_self_attn_v_proj/mean":0.00597381591796875,"train/train/tensor_act_model_layers_77_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_post_attention_layernorm/std":1.0000084425781233,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/mean":-1.5944242477416992e-06,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/max_abs":1.0073184967041016e-05,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_73_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_mlp_up_proj/std":0.225832266867428,"train/train/tensor_act_model_layers_66_mlp_up_proj/std":0.2251054773813093,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_post_attention_layernorm/mean":-0.007564544677734375,"train/train/tensor_act_model_layers_52_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_mlp_down_proj/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/mean":9.119510650634766e-06,"train/train/tensor_act_model_layers_70_self_attn_q_proj/mean":-0.0012335777282714844,"train/train/tensor_param_model_layers_23_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_79/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_v_proj/std":0.23071475356222798,"train/train/layer__model_layers_17/param/std":0.044259296630892724,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_21/act/mean":-0.010152391024998255,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/max_abs":0.08056640625,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/norm":0.0006327656039401216,"train/train/tensor_act_model_layers_53/mean":-0.0016932487487792969,"train/train/layer_model_layers_23/grad/norm":0.3031729616856916,"train/train/tensor_act_model_layers_84_post_attention_layernorm/max_abs":4.75,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model/mean":0.01849365234375,"train/train/tensor_param_model_norm_weight/mean":1,"train/train/tensor_act_model_layers_24_input_layernorm/norm":5792.560180666042,"train/train/tensor_act_model_layers_71_mlp_down_proj/norm":93.49329162981647,"train/train/tensor_act_model_layers_76_self_attn/mean":0.0002543926239013672,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/norm":0.025793618789793894,"train/train/tensor_act_model_layers_67_self_attn/mean":0.002223968505859375,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/std":6.202722796152189e-07,"train/train/tensor_act_model_layers_17_mlp_up_proj/mean":-0.00273895263671875,"train/train/tensor_param_model_layers_29_mlp_gate_proj_weight/mean":-0.00012874603271484375,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/max_abs":0.076171875,"train/train/tensor_grad_model_layers_77_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_64/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_self_attn_k_proj/std":0.2275395086390497,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/mean":2.9883813112974167e-07,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/std":3.456057961156872e-05,"train/train/layer_model_layers_73/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp_down_proj/norm":90.90465485676727,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/max_abs":0.006866455078125,"train/train/tensor_act_model_layers_44_input_layernorm/norm":5792.584228516185,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn/std":0.05041670143461329,"train/train/tensor_param_model_layers_10_mlp_gate_proj_weight/mean":-1.3768672943115234e-05,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/max_abs":0.07958984375,"train/train/tensor_act_model_layers_33/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_input_layernorm/mean":-0.0625,"train/train/tensor_act_model_layers_1_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_mlp_gate_proj_weight/std":9.814252008079036e-05,"train/train/tensor_act_model_layers_2_mlp_up_proj/std":0.22632190592002352,"train/train/tensor_act_model_layers_79_mlp_down_proj/max_abs":0.08154296875,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/mean":2.0176172256469727e-05,"train/train/tensor_act_model_layers_39_self_attn/std":0.049744708700733505,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/max_abs":0.083984375,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp_up_proj/mean":-0.0184326171875,"train/train/layer_model_layers_39/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_k_proj/mean":-0.005352020263671875,"train/train/tensor_act_model_layers_28_self_attn_o_proj/std":0.05066255488261313,"train/train/layer_model_layers_4/act/std":0.41221734851685654,"train/train/tensor_act_model_layers_62_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn_v_proj/max_abs":1.125,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_67_self_attn_v_proj/std":0.2304703628556342,"train/train/tensor_act_model_rotary_emb/norm":2297.209228515625,"train/train/tensor_act_model_layers_29_self_attn/max_abs":0.2216796875,"train/train/tensor_act_model_layers_0_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_v_proj/mean":-0.00323486328125,"train/train/tensor_param_model_layers_78_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/norm":0.02421696256378912,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/mean":-0.00011110305786132812,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/std":0.00017702822071666318,"train/train/tensor_act_model_layers_59_input_layernorm/std":1.0000102314574633,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/max_abs":0.0035858154296875,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_76/act/mean":0.002563204084123884,"train/train/tensor_param_model_layers_67_mlp_gate_proj_weight/mean":6.103515625e-05,"train/train/tensor_act_model_layers_72_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_63_self_attn_o_proj/max_abs":0.2197265625,"train/train/tensor_act_model_layers_31_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn/norm":266.3064247910867,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/mean":-0.0001373291015625,"train/train/tensor_act_model_layers_75_self_attn_v_proj/norm":1327.2976518926769,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36/max_abs":1.484375,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn/std":0.047242308874641745,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/max_abs":0.08203125,"train/train/layer_model_layers_48/grad/mean":-4.591469782987223e-08,"train/train/tensor_act_model_layers_24_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_mlp_gate_proj_weight/max_abs":0.000858306884765625,"train/train/tensor_act_model_layers_35_mlp_down_proj/mean":0.000400543212890625,"train/train/tensor_act_model_layers_31_mlp_gate_proj/norm":1839.7101716371658,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_32/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp_up_proj/mean":-0.0089263916015625,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_48/param/norm":17.928773465263873,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/norm":0.0030231373543069887,"train/train/layer_model_layers_25/grad/max_abs":0.00616455078125,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/std":0.019775390625,"train/train/layer__model_layers_18/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/mean":1.6652047634124756e-06,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/norm":0.0006431640679011618,"train/train/tensor_act_model_layers_4_self_attn_q_proj/norm":1336.9754575230368,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_51/mean":-0.001280069351196289,"train/train/layer_model_layers_24/act/norm":9007.694435705303,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/max_abs":0.00034332275390625,"train/train/tensor_param_model_layers_45_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_10/grad/mean":-3.475489647145123e-06,"train/train/tensor_act_model_layers_74_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn_o_proj/std":0.05139277155579566,"train/train/tensor_act_model_layers_53_self_attn_q_proj/max_abs":1.0546875,"train/train/tensor_act_model_layers_70_mlp_down_proj/std":0.015869989751199232,"train/train/layer_model_layers_19/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_self_attn_v_proj/max_abs":1.125,"train/train/layer__model_layers_81/param/std":0.04423440975329789,"train/train/layer_model_layers_16/act/frac_near_user_limit":0,"train/train/layer__model_layers_57/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/max_abs":0.08203125,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/norm":0.0011466893686260461,"train/train/layer_model_layers_51/act/max_abs":4.875,"train/train/tensor_act_model_layers_14_self_attn/max_abs":0.2451171875,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/norm":0.027237139361840156,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/mean":-8.23289155960083e-07,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/max_abs":0.004974365234375,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp_up_proj/mean":0.0019359588623046875,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_4/param/norm":17.937302644820235,"train/train/tensor_act_model_layers_65/max_abs":1.90625,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/norm":0.09780201523621657,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/mean":3.3266842365264893e-06,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_76_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_mlp_up_proj/std":0.2253444352207325,"train/train/tensor_param_model_layers_20_mlp_gate_proj_weight/max_abs":0.09228515625,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/std":0.00040113998451501276,"train/train/tensor_act_model_layers_57_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_post_attention_layernorm/max_abs":4.8125,"train/train/tensor_act_model_layers_37/std":0.30029895500166204,"train/train/tensor_act_model_layers_17_mlp/std":0.015564675963508922,"train/train/layer_model_layers_78/grad/mean":-6.34089376713482e-08,"train/train/tensor_act_model_layers_37_mlp_down_proj/norm":92.1369030607865,"train/train/tensor_act_model_layers_76_self_attn_k_proj/norm":1303.4384206792336,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_mlp_gate_proj_weight/mean":6.7148357629776e-07,"train/train/tensor_act_model_layers_12_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_61_self_attn_k_proj_weight/mean":0.0001468658447265625,"train/train/tensor_act_model_layers_46_mlp_gate_proj/mean":0.0059356689453125,"train/train/tensor_act_model_layers_42_self_attn/mean":6.67572021484375e-05,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/mean":-1.5124678611755371e-05,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/norm":2.53125,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/max_abs":6.079673767089844e-06,"train/train/tensor_act_model_layers_64_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_v_proj/mean":-0.0102081298828125,"train/train/tensor_act_model_layers_35_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/norm":2.5625,"train/train/layer_model_layers_60/act/mean":-0.001436659267970494,"train/train/layer__model_layers_77/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_mlp_gate_proj_weight/norm":0.0354729270996131,"train/train/tensor_act_model_layers_65_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_input_layernorm/max_abs":4.875,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_83/grad/mean":2.7705932843503243e-07,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/norm":0.0014222785549288523,"train/train/tensor_grad_model_layers_82_mlp_down_proj_weight/std":7.042435934616253e-05,"train/train/tensor_act_model_layers_63_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn/max_abs":0.2275390625,"train/train/layer_model_layers_0/act/mean":-0.0033493212291172574,"train/train/tensor_act_model_layers_50_self_attn/max_abs":0.21484375,"train/train/tensor_act_model_layers_79_self_attn_o_proj/max_abs":0.2265625,"train/train/layer__model_layers_22/param/std":0.04424867901515213,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_90_self_attn_v_proj/std":0.23169703013950754,"train/train/tensor_grad_model_layers_90_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/norm":0.001221989907108112,"train/train/layer_model_layers_92/grad/max_abs":0.0038909912109375,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/std":0.00011692388387670135,"train/train/tensor_act_model_layers_0_mlp_down_proj/mean":0.00012385845184326172,"train/train/tensor_act_model_layers_35_self_attn/max_abs":0.23046875,"train/train/tensor_act_model_layers_58_post_attention_layernorm/mean":-0.0049896240234375,"train/train/tensor_act_model_layers_76_input_layernorm/mean":0.0027561187744140625,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/mean":0.0003147125244140625,"train/train/layer_model_layers_49/act/std":0.4211858827642208,"train/train/tensor_act_model_layers_42_self_attn_v_proj/std":0.2256004418627213,"train/train/tensor_act_model_layers_65_self_attn/norm":295.60741299078444,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_mlp_gate_proj_weight/max_abs":0.0888671875,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp_up_proj/mean":-0.00400543212890625,"train/train/tensor_act_model_layers_1_post_attention_layernorm/max_abs":5.3125,"train/train/tensor_act_model_layers_37_self_attn/mean":-0.00024941563606262207,"train/train/tensor_act_model_layers_69_self_attn_k_proj/std":0.22632427526764404,"train/train/tensor_act_model_layers_57_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/mean":6.198883056640625e-05,"train/train/layer_model_layers_51/act/norm":9148.347587188906,"train/train/layer__model_layers_80/param/std":0.04424250854894052,"train/train/tensor_act_model_layers_3_self_attn/mean":-0.0012998580932617188,"train/train/tensor_param_model_layers_27_mlp_gate_proj_weight/mean":-0.0001010894775390625,"train/train/tensor_act_model_layers_47_input_layernorm/mean":-0.003047943115234375,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/max_abs":0.08056640625,"train/train/tensor_act_model_layers_39_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_60/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_19/grad/std":0.000407998706342669,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/mean":-6.914138793945312e-05,"train/train/tensor_act_model_layers_14_self_attn_o_proj/norm":283.3354800938247,"train/train/tensor_act_model_layers_4_self_attn_o_proj/mean":-0.00110626220703125,"train/train/tensor_param_model_layers_60_mlp_gate_proj_weight/norm":3.609375,"train/train/layer_model_layers_68/act/std":0.4257248399938181,"train/train/tensor_act_model_layers_18_self_attn_k_proj/mean":0.005390167236328125,"train/train/tensor_act_model_layers_18_self_attn_q_proj/max_abs":1.171875,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/norm":0.00011464835281720274,"train/train/tensor_act_model_layers_25_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/mean":9.290488378610462e-10,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/std":7.027613497268823e-05,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/max_abs":0.00153350830078125,"train/train/tensor_act_model_layers_31_input_layernorm/norm":5792.573364260265,"train/train/tensor_act_model_layers_86_self_attn_o_proj/max_abs":0.2412109375,"train/train/layer_model_layers_23/act/norm":9017.006608771067,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/std":0.0003622092707028496,"train/train/tensor_act_model_layers_41_self_attn_v_proj/std":0.22583662899929044,"train/train/tensor_act_model_layers_85_self_attn_k_proj/mean":0.0016632080078125,"train/train/tensor_act_model_layers_13_self_attn_k_proj/norm":1315.323322946825,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/mean":2.4512410163879395e-06,"train/train/tensor_act_model_layers_46_self_attn_v_proj/mean":0.00499725341796875,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/mean":0.0002727508544921875,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_mlp_gate_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_80_input_layernorm_weight/mean":-8.590519428253174e-06,"train/train/tensor_grad_model_layers_38_self_attn_v_proj_weight/norm":0.13672136576738766,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/max_abs":0.0103759765625,"train/train/layer__model_layers_93/param/max_abs":1,"train/train/tensor_act_model_layers_12_self_attn_v_proj/mean":0.0012912750244140625,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/norm":2.53125,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/std":0.00010219846997165893,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/mean":-6.389617919921875e-05,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/std":0.0005710155526478334,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/max_abs":1.2159347534179688e-05,"train/train/tensor_param_model_layers_72_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp/std":0.016296647694087132,"train/train/layer_model_layers_59/grad/norm":0.16475673878901856,"train/train/tensor_act_model_layers_51_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_91_mlp_down_proj/max_abs":0.08447265625,"train/train/layer_model_layers_10/act/std":0.41243696624761633,"train/train/tensor_act_model_layers_70_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/norm":0.004399217649895638,"train/train/tensor_act_model_layers_30_self_attn_q_proj/mean":-0.000286102294921875,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/std":4.681851954716459e-07,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/mean":9.16188582777977e-08,"train/train/tensor_act_model_layers_23_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/norm":0.0010726152720268065,"train/train/tensor_act_model_layers_66_self_attn_q_proj/mean":-0.010009765625,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_61/act/max_abs":4.65625,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_77_mlp_gate_proj/mean":0.00632476806640625,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/norm":0.0054641167564087306,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/mean":-3.826618194580078e-05,"train/train/tensor_act_model_layers_5_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_input_layernorm/mean":-0.05316162109375,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_v_proj/std":0.22631925646491988,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/std":0.00010701994195250864,"train/train/layer_model_layers_66/grad/max_abs":0.004364013671875,"train/train/tensor_act_model_layers_27_input_layernorm/norm":5792.5695800815465,"train/train/tensor_act_model_layers_37_post_attention_layernorm/mean":-0.0074920654296875,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_mlp_down_proj/max_abs":0.08056640625,"train/train/tensor_param_model_layers_37_mlp_gate_proj_weight/std":0.02001953125,"train/train/layer_model_layers_2/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_29/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_93_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_28_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_self_attn/norm":297.9201816630965,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/max_abs":1.1265277862548828e-05,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_7/param/mean":0.0015099773912831513,"train/train/tensor_act_model_layers_93_mlp_down_proj/max_abs":0.0830078125,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/mean":3.7602148950099945e-08,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/norm":0.08515049626222031,"train/train/tensor_act_model_layers_63_mlp_gate_proj/max_abs":1.0546875,"train/train/layer_model_layers_46/act/max_abs":4.875,"train/train/tensor_act_model_layers_72_mlp_up_proj/max_abs":1.0546875,"train/train/layer_model_layers_68/grad/mean":-6.8791920403469e-08,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/mean":2.2649765014648438e-05,"train/train/tensor_act_model_layers_47_self_attn_q_proj/std":0.2316984034154589,"train/train/layer_model_layers_61/grad/std":0.0002058368853924622,"train/train/tensor_act_model_layers_19_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/norm":0.02960865694247279,"train/train/tensor_act_model_layers_84_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/max_abs":0.08447265625,"train/train/tensor_act_model_layers_24/max_abs":1.140625,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/std":5.22948398806093e-07,"train/train/tensor_grad_model_layers_24_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/std":4.223893672989621e-05,"train/train/tensor_act_model_layers_51_mlp_gate_proj/std":0.22827383170221452,"train/train/layer_model_layers_63/act/std":0.4244328983325951,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/max_abs":0.07861328125,"train/train/tensor_act_model_layers_70_mlp/std":0.015869989751199232,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/std":0.00010414013473789038,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/max_abs":0.000957489013671875,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp/norm":90.84606674281565,"train/train/tensor_act_model_layers_39_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/max_abs":4.0531158447265625e-06,"train/train/tensor_act_model_layers_4_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/mean":5.936622619628906e-05,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/std":0.02001953125,"train/train/layer_model_layers_65/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/max_abs":9.918212890625e-05,"train/train/layer_model_layers_51/act/mean":-0.00184612614767892,"train/train/tensor_act_model_layers_56_post_attention_layernorm/std":1.000008523049532,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_mlp_down_proj/norm":88.89216252348889,"train/train/tensor_act_model_layers_40/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/max_abs":0.0004215240478515625,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/mean":8.717179298400879e-07,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn/norm":277.55999526932055,"train/train/tensor_param_model_layers_55_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/mean":4.518777132034302e-06,"train/train/layer_model_layers_18/act/mean":-0.010688815798078264,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/max_abs":0.000461578369140625,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_75/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/mean":5.91278076171875e-05,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/norm":0.11315331963403309,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_67_mlp_gate_proj_weight/std":7.870931489550928e-05,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/mean":-2.291926648467779e-09,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/max_abs":0.00018024444580078125,"train/train/tensor_act_model_layers_36_self_attn_o_proj/norm":297.9201816630965,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/mean":-1.605600118637085e-06,"train/train/tensor_act_model_layers_84_self_attn_v_proj/max_abs":1.0703125,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/norm":0.00012683890815944526,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/norm":0.11250539342879357,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/mean":1.6819685697555542e-06,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/max_abs":0.0810546875,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/std":0.00022405585053082795,"train/train/tensor_act_model_layers_77_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_92_mlp/norm":88.58318316199377,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/max_abs":9.179115295410156e-06,"train/train/layer_model_layers_22/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_post_attention_layernorm_weight/std":5.528590864236566e-05,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/mean":-5.520880222320557e-06,"train/train/tensor_grad_model_layers_86_mlp_gate_proj_weight/std":6.844489041056914e-05,"train/train/tensor_act_model_layers_19_mlp_gate_proj/norm":1855.7684757131012,"train/train/tensor_act_model_layers_31_post_attention_layernorm/std":1.0000007525083572,"train/train/tensor_act_model_layers_30/max_abs":1.359375,"train/train/layer_model_layers_78/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/max_abs":0.0771484375,"train/train/tensor_act_model_layers_65_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_gate_proj_weight/max_abs":0.00193023681640625,"train/train/tensor_param_model_layers_44_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_88_post_attention_layernorm/max_abs":4.65625,"train/train/tensor_act_model_layers_32_mlp/norm":88.51529036531764,"train/train/tensor_act_model_layers_3_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_input_layernorm/max_abs":4.9375,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/std":0.0005824583642785924,"train/train/tensor_act_model_layers_25_mlp_down_proj/norm":88.08356356843734,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_85_self_attn_k_proj/std":0.2319398559804066,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/norm":0.00017059644877812484,"train/train/tensor_act_model_layers_21/max_abs":1.1328125,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/max_abs":0.07763671875,"train/train/layer__model_layers_66/param/norm":17.92656052525484,"train/train/tensor_act_model_layers_50_mlp/std":0.01596221298288338,"train/train/tensor_act_model_layers_19_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_up_proj/mean":-0.0107269287109375,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/max_abs":1.2874603271484375e-05,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/mean":-5.784386303275824e-10,"train/train/tensor_act_model_layers_85_self_attn_o_proj/mean":-0.00044392049312591553,"train/train/tensor_act_model_layers_34_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_33/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/mean":-3.1087547540664673e-06,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/std":5.897080060699554e-07,"train/train/tensor_act_model_layers_26_self_attn_o_proj/std":0.0496834458975016,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_67/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn_o_proj/std":0.046936715169399845,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/max_abs":0.08251953125,"train/train/tensor_act_model_layers_66/std":0.4062584443133802,"train/train/tensor_act_model_layers_9_self_attn_v_proj/mean":0.001873016357421875,"train/train/tensor_act_model_layers_20_self_attn_q_proj/norm":1388.119920657857,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/max_abs":0.0888671875,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_o_proj/mean":0.00043451786041259766,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/norm":0.14737441195952247,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_v_proj/std":0.22120214429899862,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/max_abs":0.00067901611328125,"train/train/tensor_act_model_layers_28_self_attn_o_proj/max_abs":0.2470703125,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74/mean":0.0012073516845703125,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/mean":1.1783558875322342e-06,"train/train/tensor_act_model_layers_80_self_attn_q_proj/norm":1252.96459421981,"train/train/tensor_act_model_layers_48_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp_up_proj/norm":1856.7847414684654,"train/train/tensor_act_model_layers_17_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_post_attention_layernorm/std":1.000002803655369,"train/train/tensor_act_model_layers_27_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/mean":5.564652383327484e-08,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_mlp_gate_proj_weight/norm":0.030679562775562587,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/max_abs":0.00102996826171875,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_input_layernorm/max_abs":4.875,"train/train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/norm":0.0023111021554835438,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/mean":-3.281235694885254e-05,"train/train/tensor_act_model_layers_48_self_attn_v_proj/std":0.22485452365959932,"train/train/tensor_param_model_layers_0_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/norm":0.0007539312705646567,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/max_abs":0.004180908203125,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/mean":-6.798654794692993e-08,"train/train/layer__model_layers_25/param/mean":0.0015787669165458025,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_43_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/norm":0.08738222370111334,"train/train/tensor_act_model_layers_79_mlp_gate_proj/max_abs":1.1328125,"train/train/layer__model_layers_67/param/mean":0.001587330272156809,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/mean":-0.00011205673217773438,"train/train/layer_model_layers_4/act/norm":8928.848038290167,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/std":0.00018556960785297226,"train/train/layer_model_layers_11/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_57_self_attn_k_proj/std":0.22827735388287046,"train/train/tensor_act_model_layers_71_self_attn/frac_near_dtype_limit":0,"train/train/layer__model_layers_10/param/mean":0.0016649100412258677,"train/train/tensor_act_model_layers_61_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_mlp_gate_proj_weight/std":0.0201416015625,"train/train/layer__model_layers_69/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/mean":-9.701616363599896e-07,"train/train/tensor_act_model_layers_55_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_self_attn/std":0.04663232584854941,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/norm":0.12059953806268338,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_39_mlp_gate_proj_weight/std":0.00010390287247328606,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_85_mlp/mean":-0.0006122589111328125,"train/train/tensor_act_model_layers_5_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/max_abs":0.0859375,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/max_abs":8.165836334228516e-06,"train/train/tensor_act_model_layers_19_self_attn/max_abs":0.2451171875,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/max_abs":0.00017070770263671875,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_7/param/norm":17.93728903403814,"train/train/tensor_act_model_layers_26_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_19/act/mean":-0.010619921343667167,"train/train/layer__model_layers_6/param/std":0.044230036214133246,"train/train/tensor_act_model_layers_41_mlp_up_proj/std":0.23071582681311362,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/mean":1.7425918485969305e-09,"train/train/tensor_grad_model_layers_40_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/norm":0.0001078208319774901,"train/train/tensor_act_model_layers_85_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/std":0.00011045738333668016,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_gate_proj_weight/max_abs":0.07958984375,"train/train/layer_model_layers_72/grad/std":0.00018541222221694296,"train/train/tensor_act_model_layers_83_self_attn/std":0.04937956841370703,"train/train/tensor_act_model_layers_27_mlp_gate_proj/mean":0.003437042236328125,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/std":6.511155993987266e-05,"train/train/tensor_act_model_layers_92_input_layernorm/max_abs":4.59375,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/std":5.530320817212795e-06,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_input_layernorm_weight/mean":1,"train/train/layer__model_layers_86/param/std":0.044246817126567174,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_85/grad/std":0.00017399740945666165,"train/train/global/grad/std":0.0004499023080608568,"train/train/tensor_act_model_layers_38_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/mean":-6.366521120071411e-06,"train/train/tensor_param_model_layers_50_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/max_abs":0.0091552734375,"train/train/tensor_act_model_layers_43_mlp_gate_proj/norm":1855.306078293028,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/mean":0.0001583099365234375,"train/train/tensor_act_model_layers_70_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_k_proj/std":0.22925259964371425,"train/train/layer_model_layers_18/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/mean":-4.380941390991211e-06,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/std":0.0012933857699895877,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_26_post_attention_layernorm/mean":-0.04461669921875,"train/train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn_v_proj/mean":0.003772735595703125,"train/train/tensor_act_model_layers_65_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/std":0.0198974609375,"train/train/layer_model_layers_30/act/max_abs":4.84375,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/std":3.3061240387905444e-05,"train/train/layer__model_layers_63/param/mean":0.0015499216160052651,"train/train/tensor_grad_model_layers_55_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_24/grad/std":0.00033406426221818246,"train/train/tensor_act_model_layers_5_mlp/std":0.01562518882451105,"train/train/layer_model_layers_75/grad/std":0.0001864713839058574,"train/train/tensor_act_model_layers_61_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/std":0.00011070694463775569,"train/train/tensor_act_model_layers_44_self_attn/norm":285.6391186456364,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/max_abs":0.0004367828369140625,"train/train/tensor_act_model_layers_29_self_attn_q_proj/std":0.22973945404589227,"train/train/tensor_grad_model_layers_10_mlp_gate_proj_weight/max_abs":0.0035247802734375,"train/train/tensor_act_model_layers_64_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_mlp_gate_proj/norm":1818.6585809690068,"train/train/tensor_act_model_layers_35_mlp_up_proj/mean":0.0029201507568359375,"train/train/tensor_grad_model_layers_21_input_layernorm_weight/norm":0.003506875076967856,"train/train/tensor_grad_model_layers_39_post_attention_layernorm_weight/norm":0.0010758122632564331,"train/train/tensor_act_model_layers_80_mlp_down_proj/std":0.015533838304915061,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_q_proj/max_abs":1.2109375,"train/train/tensor_act_model_layers_16/std":0.1936094492403488,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/norm":0.00021644520515750498,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/max_abs":0.08642578125,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_5/param/max_abs":1,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/mean":1.7062120605260134e-09,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/max_abs":0.005157470703125,"train/train/tensor_grad_model_layers_49_mlp_gate_proj_weight/std":8.107999467059865e-05,"train/train/tensor_act_model_layers_34_mlp_down_proj/std":0.015671156102791833,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/std":7.526969750264878e-05,"train/train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn/mean":0.0019893646240234375,"train/train/tensor_act_model_layers_45_self_attn_v_proj/max_abs":1.1953125,"train/train/tensor_param_model_layers_37_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_49_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_1_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_56_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/max_abs":0.00014591217041015625,"train/train/tensor_act_model_layers_58_self_attn_v_proj/max_abs":1.140625,"train/train/tensor_act_model_layers_42_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/std":9.01999551205506e-05,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/max_abs":0.005523681640625,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/norm":0.10828235355932282,"train/train/layer_model_layers_19/grad/max_abs":0.007659912109375,"train/train/tensor_act_model_layers_67_self_attn_o_proj/mean":0.002223968505859375,"train/train/tensor_act_model_layers_60_post_attention_layernorm/std":1.000005957652763,"train/train/tensor_act_model_layers_43_input_layernorm/std":1.0000051789259077,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/mean":4.2980536818504333e-07,"train/train/tensor_param_model_layers_33_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/mean":5.030632019042969e-05,"train/train/tensor_act_model_layers_40_self_attn_o_proj/norm":277.3989473045707,"train/train/tensor_act_model_layers_15_post_attention_layernorm/norm":5792.528686527053,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/std":0.020263671875,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/mean":2.2794120013713837e-07,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/norm":0.001976223054441053,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/max_abs":0.0830078125,"train/train/tensor_param_model_layers_18_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_58_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25/std":0.2456119770643835,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_input_layernorm/std":1.0000041182999446,"train/train/tensor_grad_model_layers_76_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/std":0.00012602482695702215,"train/train/tensor_act_model_layers_48_self_attn_v_proj/mean":0.0096588134765625,"train/train/tensor_act_model_layers_58_mlp_up_proj/max_abs":1.1328125,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/std":4.716119509845209e-07,"train/train/tensor_act_model_layers_17_mlp_down_proj/mean":0.0005807876586914062,"train/train/layer_model_layers_34/act/max_abs":5.0625,"train/train/layer_model_layers_27/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_k_proj/norm":1308.9059244243554,"train/train/tensor_act_model_layers_87_mlp/norm":89.33651009631114,"train/train/global/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_gate_proj/norm":1851.9391085830941,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_79_mlp_up_proj/norm":1845.5123532883047,"train/train/tensor_act_model_layers_17_mlp/norm":90.31826523593222,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_post_attention_layernorm/mean":-0.0011548101902008057,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_23_self_attn/max_abs":0.212890625,"train/train/tensor_act_model_layers_6_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_13_mlp_up_proj/mean":-0.001789093017578125,"train/train/tensor_act_model_layers_66_self_attn_q_proj/std":0.2260843814353526,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/mean":-0.00021457672119140625,"train/train/tensor_act_model_layers_85_mlp/max_abs":0.08447265625,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/max_abs":0.0230712890625,"train/train/tensor_param_model_layers_57_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_v_proj/mean":0.01361083984375,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/max_abs":0.07568359375,"train/train/tensor_act_model_layers_35_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/max_abs":0.08447265625,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_q_proj/max_abs":1.1015625,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/mean":-1.7113052308559418e-08,"train/train/tensor_grad_model_layers_8_mlp_gate_proj_weight/max_abs":0.004119873046875,"train/train/tensor_act_model_layers_68_self_attn_q_proj/mean":-0.005859375,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_input_layernorm/mean":0.0115509033203125,"train/train/tensor_act_model_layers_67_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_gate_proj_weight/max_abs":0.08740234375,"train/train/layer_model_layers_57/act/norm":9159.346477825648,"train/train/tensor_act_model_layers_91_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_k_proj/max_abs":1.078125,"train/train/tensor_param_model_layers_18_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_19_mlp_down_proj/norm":89.07891139170341,"train/train/tensor_act_model_layers_19_self_attn_k_proj/norm":1333.8456487103515,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/max_abs":0.00156402587890625,"train/train/tensor_param_model_layers_29_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_92/grad/norm":0.13659445209117899,"train/train/tensor_act_model_layers_49/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/norm":0.1198461933707542,"train/train/tensor_act_model_layers_36_mlp_down_proj/mean":0.00035858154296875,"train/train/tensor_act_model_layers_57_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_o_proj/mean":-0.0022430419921875,"train/train/tensor_param_model_layers_93_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/mean":-1.0669231414794922e-05,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/norm":0.0008051993734999157,"train/train/layer_model_layers_2/grad/std":0.0013900125271857535,"train/train/tensor_act_model_layers_88_self_attn_k_proj/mean":0.0080108642578125,"train/train/tensor_grad_model_layers_2_mlp_gate_proj_weight/max_abs":0.005035400390625,"train/train/tensor_grad_model_layers_20_mlp_gate_proj_weight/std":0.00015103276812303133,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_mlp_down_proj/std":0.01568770068375377,"train/train/tensor_act_model_layers_71_post_attention_layernorm/norm":5792.592773440319,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_33/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_post_attention_layernorm/std":1.000001062078245,"train/train/tensor_grad_model_layers_88_mlp_gate_proj_weight/norm":0.02350875830276918,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_33_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_norm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/norm":0.0020380089312995193,"train/train/tensor_param_model_layers_14_mlp_gate_proj_weight/max_abs":0.09716796875,"train/train/tensor_act_model_layers_21_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/max_abs":0.08447265625,"train/train/layer__model_layers_11/param/mean":0.0016325784733812448,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/norm":0.09970447264255099,"train/train/tensor_act_model_layers_41_self_attn_k_proj/norm":1322.0609459019852,"train/train/layer_model_layers_76/act/norm":9245.468880685605,"train/train/layer__model_layers_83/param/max_abs":1,"train/train/tensor_grad_model_layers_7_self_attn_q_proj_weight/max_abs":1.919269561767578e-05,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/mean":-1.9837170839309692e-07,"train/train/tensor_act_model_layers_89_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_v_proj/mean":0.01409912109375,"train/train/tensor_act_model_layers_37_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/mean":2.88418959826231e-08,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_59_post_attention_layernorm/max_abs":4.53125,"train/train/tensor_act_model_layers_49_self_attn_o_proj/std":0.04968509503870688,"train/train/tensor_act_model_layers_49_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp/max_abs":0.08837890625,"train/train/layer_model_layers_65/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/std":3.5343414247517524e-05,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/std":6.881549967163679e-05,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_k_proj/mean":0.0135040283203125,"train/train/tensor_param_model_layers_87_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/mean":-1.8961727619171143e-06,"train/train/tensor_act_model_layers_51_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/mean":0.00016498565673828125,"train/train/tensor_act_model_layers_92_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp_gate_proj/max_abs":1.078125,"train/train/tensor_grad_model_layers_42_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/std":0.0002293668400698939,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/max_abs":0.00225830078125,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/std":0.0004776606018374101,"train/train/tensor_grad_model_layers_91_mlp_down_proj_weight/std":7.369642539989618e-05,"train/train/tensor_act_model_layers_38_mlp_down_proj/mean":0.00017213821411132812,"train/train/tensor_act_model_layers_80_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_self_attn_k_proj/max_abs":1.1953125,"train/train/tensor_param_model_layers_54_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_down_proj/norm":90.49846609600961,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_mlp_gate_proj_weight/mean":-7.066410034894943e-08,"train/train/layer__model_layers_81/param/norm":17.92654690631676,"train/train/tensor_act_model_layers_81_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/std":0.0011003249900568956,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/norm":0.0007450949469018138,"train/train/tensor_act_model_layers_53_post_attention_layernorm/norm":5792.589477539889,"train/train/tensor_act_model_layers_80_self_attn_q_proj/std":0.21631679454783564,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_mlp_gate_proj_weight/max_abs":0.08984375,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_79/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_o_proj/mean":8.71419906616211e-05,"train/train/tensor_param_model_layers_0_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/std":8.252866832575038e-05,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/std":0.001743885943184206,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_79_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn_q_proj/max_abs":1.09375,"train/train/tensor_param_model_layers_84_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_23_post_attention_layernorm/mean":-0.04876708984375,"train/train/layer_model_layers_30/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_q_proj/std":0.22119498186656902,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/mean":-4.029273986816406e-05,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/mean":-4.839897155761719e-05,"train/train/tensor_param_model_layers_30_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/max_abs":0.09326171875,"train/train/tensor_param_model_layers_83_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn/std":0.05023516609611818,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/mean":-2.4410837795585394e-09,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/std":8.304432992364313e-05,"train/train/layer__model_layers_68/param/std":0.044264218713087264,"train/train/tensor_act_model_layers_55/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_k_proj/max_abs":1.0546875,"train/train/layer_model_layers_48/grad/frac_near_user_limit":0,"train/train/layer_model_layers_70/act/max_abs":4.71875,"train/train/layer_model_layers_69/grad/mean":1.7192018583021744e-07,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/norm":0.00012547637078673426,"train/train/tensor_grad_model_layers_61_mlp_up_proj_weight/norm":0.026342097460228985,"train/train/tensor_act_model_layers_45_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/max_abs":0.00185394287109375,"train/train/tensor_param_model_layers_62_mlp_gate_proj_weight/mean":1.3709068298339844e-05,"train/train/tensor_param_model_layers_52_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_62_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_26_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn/max_abs":0.23046875,"train/train/tensor_param_model_layers_86_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/max_abs":0.07470703125,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_32_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_87_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_46_mlp/std":0.01539644150568739,"train/train/tensor_act_model_layers_52_mlp/norm":89.69475723622551,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/max_abs":0.000156402587890625,"train/train/tensor_act_model_layers_58_self_attn_v_proj/mean":-0.00337982177734375,"train/train/tensor_act_model_layers_18_self_attn_o_proj/mean":-0.00017333030700683594,"train/train/tensor_act_model_layers_7_post_attention_layernorm/norm":5792.398071299964,"train/train/tensor_param_model_layers_18_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_71_self_attn_o_proj/norm":291.0904437653292,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/max_abs":6.377696990966797e-06,"train/train/tensor_act_model_layers_10_self_attn_v_proj/std":0.221926050206507,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn_q_proj/mean":0.01641845703125,"train/train/tensor_param_model_layers_43_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_64_mlp_gate_proj_weight/max_abs":0.001953125,"train/train/tensor_act_model_layers_75_input_layernorm/std":1.000000904173749,"train/train/layer__model_layers_91/param/mean":0.0015963764160918,"train/train/tensor_act_model_layers_59/norm":2210.7292971270185,"train/train/tensor_act_model_layers_65_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_q_proj/std":0.22558844913801898,"train/train/tensor_act_model_layers_86_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/max_abs":0.00017833709716796875,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/max_abs":1.0013580322265625e-05,"train/train/tensor_param_model_layers_49_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_86/act/max_abs":4.75,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn_v_proj/max_abs":1.1328125,"train/train/tensor_param_model_layers_70_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/mean":-4.495377652347088e-07,"train/train/tensor_act_model_layers_61_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/mean":-6.151199340820312e-05,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_75_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/mean":2.6106834411621094e-05,"train/train/tensor_act_model_layers_92_input_layernorm/std":1.0000033344785038,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_58/param/norm":17.93458028284674,"train/train/tensor_act_model_layers_1_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_91_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/max_abs":5.781650543212891e-06,"train/train/tensor_act_model_layers_90_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_30_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_v_proj/mean":0.027008056640625,"train/train/layer_model_layers_27/act/max_abs":4.6875,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/norm":0.02804942308077407,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/max_abs":0.000392913818359375,"train/train/layer__model_layers_51/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_post_attention_layernorm/mean":-0.04815673828125,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_v_proj/max_abs":1.140625,"train/train/tensor_act_model_layers_54_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/mean":3.762543201446533e-06,"train/train/tensor_act_model_layers_86_mlp_up_proj/max_abs":1.2265625,"train/train/tensor_act_model_layers_27_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/norm":0.12195404619336978,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_21/act/norm":8986.658167667756,"train/train/tensor_act_model_layers_20_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_mlp_gate_proj_weight/max_abs":0.0869140625,"train/train/layer_model_layers_29/act/mean":-0.007388302258082798,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/mean":9.462237358093262e-07,"train/train/tensor_act_model_layers_31_mlp_gate_proj/max_abs":1.2265625,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/std":0.00010926205463962073,"train/train/tensor_param_model_layers_55_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/mean":1.0959411156363785e-10,"train/train/tensor_act_model_layers_41_self_attn_v_proj/max_abs":1.0625,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/norm":0.028202865683930014,"train/train/tensor_param_model_layers_17_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/mean":-7.295608520507812e-05,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/norm":0.0002506680830003328,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/mean":-0.00011587142944335938,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/norm":0.0008559873816826511,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/norm":5792.55163574462,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/mean":4.9591064453125e-05,"train/train/tensor_act_model_layers_0_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_v_proj/std":0.22755061425204276,"train/train/tensor_act_model_layers_32_self_attn/norm":273.99505625892175,"train/train/tensor_act_model_layers_48_mlp/max_abs":0.08544921875,"train/train/tensor_act_model_layers_89_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_34/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/act/mean":-0.00949669097151075,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/std":0.00039323360663741835,"train/train/tensor_act_model_layers_41_post_attention_layernorm/mean":-0.009307861328125,"train/train/tensor_grad_model_layers_88_mlp_gate_proj_weight/mean":1.752050593495369e-07,"train/train/layer_model_layers_68/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_31/act/norm":9034.85411664079,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/max_abs":0.0005035400390625,"train/train/layer_model_layers_17/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_89/grad/norm":0.12909900638225563,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/max_abs":0.08447265625,"train/train/tensor_act_model_layers_22_mlp_gate_proj/norm":1857.862769772784,"train/train/tensor_act_model_layers_75_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/norm":0.03024678441367234,"train/train/layer_model_layers_66/grad/std":0.0002036440938535622,"train/train/tensor_act_model_layers_39_self_attn_o_proj/std":0.049744708700733505,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_k_proj/mean":0.0087738037109375,"train/train/tensor_act_model_layers_31_mlp/std":0.01577782823847169,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/max_abs":0.003997802734375,"train/train/tensor_act_model_layers_43_self_attn/std":0.047671234442833124,"train/train/tensor_param_model_layers_10_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_40/grad/std":0.00023323006688812475,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/max_abs":0.0947265625,"train/train/tensor_grad_model_layers_33_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_43/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/norm":0.030083769726868218,"train/train/tensor_act_model_layers_8_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_post_attention_layernorm/norm":5792.571899414797,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75/std":0.4340875527697643,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/mean":-0.000164031982421875,"train/train/tensor_act_model_layers_1_self_attn_o_proj/mean":-0.0010232925415039062,"train/train/tensor_act_model_layers_14_mlp_gate_proj/max_abs":1.2109375,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/max_abs":0.08349609375,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/std":0.019775390625,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/mean":-5.6743621826171875e-05,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_53_post_attention_layernorm/std":1.0000106906729056,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89/norm":2760.862539287569,"train/train/layer__model_layers_28/param/mean":0.0014958136167243564,"train/train/tensor_act_model_layers_0_self_attn_k_proj/norm":1320.7239630555187,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_58/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_89_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn_o_proj/std":0.049135418958793954,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/max_abs":4.947185516357422e-06,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/max_abs":0.078125,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_52_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/std":5.198786174610749e-07,"train/train/tensor_act_model_layers_22_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/mean":7.62939453125e-05,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/max_abs":0.07861328125,"train/train/tensor_act_model_layers_22_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/max_abs":0.09326171875,"train/train/tensor_act_model_layers_33/max_abs":1.4765625,"train/train/tensor_param_model_layers_81_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_73_mlp_up_proj/max_abs":1.0234375,"train/train/tensor_param_model_layers_52_mlp_gate_proj_weight/mean":2.300739288330078e-05,"train/train/layer_model_layers_63/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16/mean":-0.012298583984375,"train/train/tensor_param_model_layers_15_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/mean":1.0609626770019531e-05,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/max_abs":0.004638671875,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/mean":-0.00015163421630859375,"train/train/tensor_act_model_layers_30_mlp_down_proj/max_abs":0.08251953125,"train/train/tensor_act_model_layers_75_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/std":6.774613239371562e-05,"train/train/tensor_act_model_layers_7_self_attn_k_proj/max_abs":1.171875,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/mean":-0.000278472900390625,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/max_abs":0.005218505859375,"train/train/tensor_act_model_layers_75_self_attn_v_proj/max_abs":1.015625,"train/train/tensor_act_model_layers_60_self_attn_o_proj/norm":285.08397855122234,"train/train/layer__model_layers_33/param/frac_near_dtype_limit":0,"train/train/layer__model_layers_76/param/norm":17.930087481921973,"train/train/tensor_act_model_layers_1_self_attn_q_proj/norm":1313.3745226043125,"train/train/tensor_act_model_layers_21_mlp_down_proj/norm":89.53318284699633,"train/train/tensor_act_model_layers_43_self_attn_k_proj/max_abs":1.171875,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_90_post_attention_layernorm/norm":5792.5897216812755,"train/train/layer__model_layers_21/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/norm":0.035732422208616596,"train/train/tensor_act_model_layers_83_self_attn_v_proj/std":0.22681280144165955,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_81/act/mean":0.001270294189453125,"train/train/tensor_act_model_layers_20_self_attn_k_proj/norm":1299.791211956617,"train/train/tensor_grad_model_layers_59_mlp_gate_proj_weight/std":8.441915907142597e-05,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/mean":-7.361173629760742e-06,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_mlp_gate_proj/std":0.23242692280745442,"train/train/tensor_act_model_layers_29_self_attn_o_proj/std":0.047792070188501366,"train/train/tensor_act_model_layers_71_mlp_up_proj/std":0.22900493712198527,"train/train/tensor_act_model_layers_57_self_attn/norm":280.3629927107138,"train/train/tensor_act_model_layers_0_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_v_proj/max_abs":1.140625,"train/train/tensor_param_model_layers_50_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/mean":-8.940696716308594e-07,"train/train/tensor_act_model_layers_88_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_k_proj/norm":1325.19553900265,"train/train/tensor_grad_model_layers_13_input_layernorm_weight/mean":1.843273639678955e-05,"train/train/layer_model_layers_52/act/mean":-0.00023259435381208147,"train/train/tensor_act_model_layers_12_mlp/max_abs":0.0947265625,"train/train/tensor_param_model_layers_66_self_attn_v_proj_weight/max_abs":0.09228515625,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/mean":-1.1879019439220428e-06,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/mean":-4.101544618606567e-06,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/norm":0.12523081450181242,"train/train/tensor_act_model_layers_29_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_78/act/std":0.42779285809056916,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_mlp_gate_proj_weight/mean":-1.1324882507324219e-05,"train/train/tensor_act_model_layers_93_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_post_attention_layernorm/mean":-0.083984375,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_gate_proj_weight/norm":0.09675425726619631,"train/train/tensor_act_model_layers_52_self_attn_q_proj/max_abs":1.1015625,"train/train/tensor_param_model_layers_55_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_9_self_attn_k_proj/mean":-0.014556884765625,"train/train/layer_model_layers_91/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/norm":2.59375,"train/train/tensor_act_model_layers_57_mlp/mean":-6.763637065887451e-05,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/max_abs":1.2695789337158203e-05,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_post_attention_layernorm/norm":5792.5996093778085,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/std":0.0006419303911542727,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/norm":2.5625,"train/train/layer__model_layers_30/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_48_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/max_abs":0.00054931640625,"train/train/tensor_param_model_layers_28_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_75_post_attention_layernorm/max_abs":4.75,"train/train/tensor_act_model_layers_64_post_attention_layernorm/norm":5792.599853521403,"train/train/tensor_act_model_layers_92_self_attn/mean":0.0022830963134765625,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/mean":5.002220859751105e-10,"train/train/tensor_param_model_layers_33_input_layernorm_weight/mean":1,"train/train/layer_model_layers_47/grad/norm":0.1832134391083944,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/mean":-3.416091203689575e-06,"train/train/tensor_act_model_layers_44_mlp/norm":93.35886087081747,"train/train/tensor_param_model_layers_38_mlp_gate_proj_weight/norm":3.671875,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/mean":-5.91278076171875e-05,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_47_mlp_up_proj/mean":-0.002155303955078125,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/norm":0.00011569026630675605,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_66_self_attn_o_proj/std":0.04895238454512014,"train/train/tensor_param_model_norm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn/norm":294.34059773673096,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp_down_proj/std":0.015564372721170703,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/max_abs":0.08203125,"train/train/tensor_param_model_layers_86_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_2/act/norm":8899.041056461767,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_mlp_gate_proj_weight/std":0.0001680999897257627,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_1_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/max_abs":0.09033203125,"train/train/tensor_act_model_layers_19_post_attention_layernorm/max_abs":4.3125,"train/train/tensor_act_model_layers_45_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_self_attn/std":0.05090791166419868,"train/train/tensor_act_model_layers_87_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_up_proj/norm":1866.4789269800535,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/norm":0.030973883533446267,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/max_abs":0.0262451171875,"train/train/tensor_param_model_layers_53_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_43_self_attn/max_abs":0.24609375,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/max_abs":0.0771484375,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/max_abs":0.07861328125,"train/train/layer_model_layers_47/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_13/grad/max_abs":0.011474609375,"train/grad_norm":0.2412109375,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_35/grad/max_abs":0.006317138671875,"train/train/tensor_act_model_layers_73_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/norm":2.5625,"train/train/layer_model_layers_48/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp/mean":-0.0003795623779296875,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/max_abs":0.004150390625,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/mean":2.6941299438476562e-05,"train/train/tensor_act_model_layers_12_mlp_up_proj/std":0.224854093128335,"train/train/tensor_act_model_layers_87_self_attn_q_proj/mean":0.009429931640625,"train/train/tensor_act_model_layers_83_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_post_attention_layernorm/norm":5792.593261719676,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_68/norm":2407.825462206954,"train/train/layer_model_layers_5/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_input_layernorm/std":1.0000027916700815,"train/train/tensor_act_model_layers_85_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/max_abs":0.0024261474609375,"train/train/layer_model_layers_20/grad/std":0.00040358736340884724,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/norm":0.11205575970166869,"train/train/tensor_act_model_layers_59_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/mean":4.7404319047927856e-07,"train/train/tensor_grad_model_layers_42_post_attention_layernorm_weight/max_abs":0.00018596649169921875,"train/train/tensor_act_model_layers_86/mean":0.0070343017578125,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/max_abs":0.080078125,"train/train/layer__model_layers_65/param/max_abs":1,"train/train/layer_model_layers_3/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_39/act/norm":9088.769846996956,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/std":7.917866739710582e-05,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_64_self_attn_k_proj/std":0.2263290763835972,"train/train/tensor_act_model_layers_46_mlp_down_proj/mean":-7.270276546478271e-05,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_o_proj_weight/max_abs":0.0123291015625,"train/train/tensor_param_model_layers_37_mlp_gate_proj_weight/max_abs":0.083984375,"train/train/tensor_act_model_layers_2_self_attn_k_proj/mean":-0.003093719482421875,"train/train/tensor_param_model_layers_73_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_23_mlp/frac_near_user_limit":0,"train/train/layer__model_layers_15/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/norm":0.048739849403189386,"train/train/tensor_act_model_layers_49_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_gate_proj_weight/max_abs":0.08984375,"train/train/tensor_act_model_layers_29_mlp_up_proj/std":0.22827455651159034,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/norm":0.00013781666301458233,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/max_abs":0.0908203125,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/mean":-3.5015546018257737e-09,"train/train/tensor_act_model_layers_7_self_attn_q_proj/norm":1314.5418707118815,"train/train/tensor_act_model_layers_35_mlp_gate_proj/mean":0.0021839141845703125,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/max_abs":0.08544921875,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/max_abs":0.0791015625,"train/train/tensor_act_model_layers_90_post_attention_layernorm/std":1.0000037591600914,"train/train/tensor_act_model_layers_56_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_mlp_up_proj_weight/norm":0.025398693269998277,"train/train/tensor_act_model_layers_69_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_mlp/mean":-0.0003209114074707031,"train/train/tensor_act_model_layers_22/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_post_attention_layernorm/max_abs":4.625,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_self_attn_q_proj/std":0.22558836309443073,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_50/act/max_abs":4.8125,"train/train/tensor_act_model_layers_5_mlp_up_proj/std":0.22802963245476707,"train/train/tensor_grad_model_layers_93_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_71/grad/mean":4.220600261726357e-08,"train/train/tensor_act_model_layers_84/max_abs":2.234375,"train/train/layer_model_layers_73/grad/norm":0.1576208969387318,"train/train/tensor_act_model_layers_61_self_attn_o_proj/norm":281.2532324693673,"train/train/tensor_act_model_layers_19_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/max_abs":0.00127410888671875,"train/train/tensor_grad_model_layers_19_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/max_abs":1.2421875,"train/train/layer_model_layers_78/grad/std":0.0001889251456341701,"train/train/tensor_act_model_layers_26_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/max_abs":0.001495361328125,"train/train/layer_model_layers_89/grad/max_abs":0.004119873046875,"train/train/tensor_param_model_norm_weight/std":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_38/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/max_abs":0.080078125,"train/train/tensor_act_model_layers_4/norm":472.2756790909761,"train/train/tensor_param_model_layers_78_self_attn_q_proj_weight/mean":0.00013446807861328125,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_93_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_post_attention_layernorm/std":0.9980529572213833,"train/train/tensor_param_model_layers_3_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/std":3.7638113362018134e-07,"train/train/tensor_param_model_layers_79_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/mean":-5.995843821438029e-10,"train/train/layer__model_layers_47/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/max_abs":0.004119873046875,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/max_abs":0.000598907470703125,"train/train/tensor_act_model_layers_0_self_attn_v_proj/max_abs":1.0234375,"train/train/layer_model_layers_23/act/max_abs":4.78125,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/norm":0.3320355583360187,"train/train/tensor_act_model_layers_19_self_attn_q_proj/norm":1277.0468386488456,"train/train/tensor_act_model_layers_31/std":0.2749072594482632,"train/train/tensor_act_model_layers_2_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/mean":5.918554961681366e-07,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/norm":0.024492214999875535,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/max_abs":0.080078125,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/max_abs":0.004730224609375,"train/train/tensor_act_model_layers_25/norm":1426.1336637008128,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/norm":50.61179826723799,"train/train/tensor_grad_model_layers_41_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_self_attn_v_proj_weight/mean":-9.250640869140625e-05,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/norm":0.00011679566265404655,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/norm":0.1046022210157053,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/mean":6.957634468562901e-10,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/max_abs":0.08837890625,"train/train/tensor_grad_model_layers_63_mlp_gate_proj_weight/mean":1.0794028639793396e-06,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/mean":5.275069270282984e-11,"train/train/tensor_param_model_layers_78_mlp_gate_proj_weight/max_abs":0.080078125,"train/train/tensor_act_model_layers_65_self_attn_q_proj/norm":1313.3486275304178,"train/train/tensor_act_model_layers_24_post_attention_layernorm/max_abs":4.6875,"train/train/tensor_param_model_layers_54_mlp_gate_proj_weight/mean":7.152557373046875e-05,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/norm":0.00012640232661385596,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_self_attn_v_proj/max_abs":1.1171875,"train/train/layer_model_layers_60/grad/max_abs":0.005279541015625,"train/train/tensor_grad_model_layers_9_self_attn_q_proj_weight/mean":-1.1839347280329093e-08,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_post_attention_layernorm/norm":5792.599609378138,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/norm":0.010180482931378233,"train/train/tensor_act_model_layers_72_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn/mean":-0.00017333030700683594,"train/train/layer__model_layers_7/param/max_abs":1,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp/mean":-0.00012159347534179688,"train/train/tensor_param_model_layers_0_mlp_gate_proj_weight/mean":-0.00012969970703125,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/norm":0.00048272220587677563,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/mean":-5.030632019042969e-05,"train/train/tensor_act_model_layers_73_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp_down_proj/norm":91.92082456292871,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/norm":0.0043155405635017935,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_6_mlp/max_abs":0.0849609375,"train/train/tensor_act_model_layers_35/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_45_self_attn_o_proj/norm":291.7028413964522,"train/train/tensor_act_model_layers_91_mlp_gate_proj/std":0.22754096881739155,"train/train/tensor_act_model_layers_37_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/max_abs":0.00469970703125,"train/train/tensor_act_model_layers_37_self_attn_o_proj/norm":279.51952065436603,"train/train/layer_model_layers_51/grad/mean":-1.9776184466346377e-07,"train/train/tensor_act_model_layers_32_mlp_gate_proj/max_abs":1.1484375,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/norm":0.00010514097147618353,"train/train/tensor_act_model_layers_70_self_attn_o_proj/max_abs":0.2421875,"train/train/tensor_act_model_layers_79_self_attn_k_proj/mean":0.0017232894897460938,"train/train/tensor_act_model_layers_40_mlp/std":0.015336082909851638,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/mean":-2.608430804684758e-09,"train/train/tensor_act_model_layers_77_mlp_down_proj/mean":0.0003638267517089844,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/max_abs":0.00146484375,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/mean":-4.563480615615845e-06,"train/train/tensor_act_model_layers_92_self_attn_o_proj/norm":294.7013685816614,"train/train/tensor_act_model_layers_29_self_attn_v_proj/norm":1307.0789899349604,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp_down_proj/max_abs":0.0830078125,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/std":7.341004848161449e-05,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/max_abs":0.09228515625,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/max_abs":1.1682510375976562e-05,"train/train/tensor_act_model_layers_70_self_attn_v_proj/std":0.22559284095398371,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_2/max_abs":0.3671875,"train/train/tensor_act_model_layers_25_mlp/mean":0.0006303787231445312,"train/train/tensor_act_model_layers_29_mlp_down_proj/max_abs":0.083984375,"train/train/layer_model_layers_67/grad/mean":1.0720262007552451e-07,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/std":9.215004290171237e-07,"train/train/tensor_param_model_layers_12_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_71/max_abs":1.9921875,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_76/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp/std":0.015213348939213412,"train/train/layer_model_layers_1/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp_down_proj/max_abs":0.07861328125,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_q_proj/max_abs":1.171875,"train/train/tensor_grad_model_layers_9_mlp_gate_proj_weight/std":0.0002090190209492076,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_38_mlp_gate_proj/std":0.22631993965611322,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_mlp_gate_proj/norm":1842.0237504932002,"train/train/tensor_act_model_layers_88_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/norm":0.0009629466383167716,"train/train/tensor_param_model_layers_23_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/max_abs":0.00738525390625,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_41_self_attn_q_proj/norm":1307.5623721755126,"train/train/tensor_param_model_layers_41_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_74/max_abs":2.125,"train/train/layer_model_layers_6/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_0/act/norm":8886.55695254476,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_o_proj/mean":0.00026726722717285156,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/mean":-8.835922926664352e-08,"train/train/tensor_act_model_layers_26_self_attn_q_proj/std":0.2226641197654233,"train/train/tensor_act_model_layers_33_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_self_attn_q_proj_weight/std":0.019775390625,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/mean":-4.482269287109375e-05,"train/train/layer_model_layers_89/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_mlp_up_proj_weight/max_abs":0.00153350830078125,"train/train/tensor_param_model_layers_7_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_7_self_attn_v_proj/std":0.22559692874157905,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_52_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_post_attention_layernorm/max_abs":4.8125,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_mlp_up_proj/std":0.22827279417534266,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/norm":9.011402294658314e-05,"train/train/tensor_param_model_layers_44_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/mean":2.068190951831639e-09,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/max_abs":0.08642578125,"train/train/tensor_grad_model_layers_75_mlp_gate_proj_weight/max_abs":0.0013580322265625,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/norm":0.4464343479327729,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/max_abs":0.00408935546875,"train/train/tensor_act_model_layers_7_self_attn_o_proj/mean":0.000713348388671875,"train/train/layer_model_layers_25/act/max_abs":4.71875,"train/train/tensor_act_model_layers_16_post_attention_layernorm/std":1.0000030603212975,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_gate_proj_weight/mean":-6.4849853515625e-05,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/max_abs":0.00019359588623046875,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/mean":-3.234599716961384e-07,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/norm":0.11036317477273272,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/max_abs":0.08203125,"train/train/layer_model_layers_37/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/max_abs":0.0003986358642578125,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/norm":0.00018943882737240442,"train/train/tensor_act_model_layers_8_self_attn_v_proj/mean":-0.019012451171875,"train/train/tensor_param_model_layers_5_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/mean":-1.1117663234472275e-08,"train/train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_mlp_up_proj/std":0.22803019186422213,"train/train/tensor_act_model_layers_40_mlp_down_proj/std":0.015336082909851638,"train/train/tensor_act_model_layers_32_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/std":0.003420881166795853,"train/train/tensor_act_model_layers_43_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_mlp_down_proj/std":0.015442409488830584,"train/train/tensor_grad_model_layers_1_mlp_gate_proj_weight/mean":-9.266659617424011e-07,"train/train/tensor_act_model_layers_32_self_attn_v_proj/norm":1273.6854476251417,"train/train/layer_model_layers_55/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_k_proj/max_abs":1.078125,"train/train/tensor_act_model_layers_45_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/mean":-2.9802322387695312e-05,"train/train/tensor_act_model_layers_91_input_layernorm/norm":5792.605468764252,"train/train/tensor_act_model_layers_63_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_65/param/norm":17.943644145891184,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_input_layernorm/max_abs":4.65625,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/norm":0.00012072882176090028,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_mlp_gate_proj_weight/norm":0.026586188244260744,"train/train/tensor_param_model_layers_79_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_8_self_attn_o_proj_weight/max_abs":0.08349609375,"train/train/tensor_act_model_layers_25_self_attn_o_proj/max_abs":0.26171875,"train/train/layer_model_layers_43/grad/mean":1.904655461506148e-07,"train/train/tensor_act_model_layers_78/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/mean":-1.6443664208054543e-09,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/mean":-1.1846423149108887e-05,"train/train/tensor_act_model_layers_82_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_q_proj/mean":0.0020742416381835938,"train/train/layer_model_layers_18/grad/std":0.00039141981169272725,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_gate_proj_weight/max_abs":0.08056640625,"train/train/layer_model_layers_55/grad/std":0.0002013889744952472,"train/train/tensor_param_model_layers_52_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_10/grad/norm":0.4265074510543661,"train/train/layer_model_layers_47/grad/std":0.00022617604079657456,"train/train/tensor_act_model_layers_88_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_mlp/norm":89.53318284699633,"train/train/tensor_act_model_layers_75_input_layernorm/norm":5792.598754888168,"train/train/tensor_act_model_layers_90_mlp_up_proj/std":0.2246114587086862,"train/train/tensor_act_model_layers_70_self_attn_k_proj/mean":-0.004261016845703125,"train/train/tensor_act_model_layers_65_mlp_gate_proj/std":0.22729675200552205,"train/train/layer__model_layers_12/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_46_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_mlp_down_proj/mean":0.0006856918334960938,"train/train/tensor_act_model_layers_39_mlp_gate_proj/std":0.22436936450431946,"train/train/tensor_act_model_layers_4_self_attn/mean":-0.00110626220703125,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_mlp_gate_proj/max_abs":1.140625,"train/train/tensor_param_model_layers_11_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/max_abs":5.9604644775390625e-06,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/norm":0.0009957017547984103,"train/train/tensor_grad_model_layers_23_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_q_proj/mean":0.0144805908203125,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/mean":-0.00020503997802734375,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/max_abs":0.09423828125,"train/train/tensor_act_model_layers_68_self_attn_o_proj/mean":0.0004940032958984375,"train/train/layer_model_layers_0/act/max_abs":5.25,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/mean":3.6013716453453526e-09,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_80_mlp/std":0.015533838304915061,"train/train/layer_model_layers_66/act/mean":-0.002912414925439017,"train/train/tensor_act_model_layers_33_mlp_down_proj/std":0.014984534313884093,"train/train/tensor_act_model_layers_37/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_0_input_layernorm/mean":-0.001682281494140625,"train/train/tensor_act_model_layers_62_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/std":9.706010854559219e-05,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/norm":9.943455430488552e-05,"train/train/tensor_act_model_layers_46_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_post_attention_layernorm/mean":-0.05230712890625,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/max_abs":0.00439453125,"train/train/layer_model_layers_93/grad/frac_near_user_limit":0,"train/train/layer__model_layers_15/param/mean":0.0015783919931015992,"train/train/tensor_grad_model_layers_62_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/max_abs":0.004180908203125,"train/train/tensor_act_model_layers_20_input_layernorm/max_abs":4.34375,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/mean":3.11434268951416e-06,"train/train/tensor_param_model_layers_30_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/norm":2.5625,"train/train/layer_model_layers_55/grad/mean":6.824135402574925e-08,"train/train/tensor_act_model_layers_5_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn/max_abs":0.240234375,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/mean":-0.0002384185791015625,"train/train/tensor_act_model_layers_32_mlp/max_abs":0.0810546875,"train/train/tensor_act_model_layers_77_self_attn_k_proj/max_abs":1.03125,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/norm":0.3052456951165848,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/std":8.752299892674753e-05,"train/train/layer__model_layers_16/param/mean":0.0015138099420461194,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/mean":-1.1488795280456543e-05,"train/train/tensor_act_model_layers_82_self_attn/max_abs":0.2412109375,"train/train/layer__model_layers_45/param/mean":0.0014859317059449956,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_58_self_attn_k_proj_weight/std":4.866479528221032e-07,"train/train/layer__model_layers_87/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp/std":0.016205620022610872,"train/train/tensor_act_model_layers_11_self_attn/max_abs":0.236328125,"train/train/tensor_param_model_layers_69_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_q_proj/norm":1301.1707309783735,"train/train/tensor_act_model_layers_58_self_attn_k_proj/norm":1309.1433454193339,"train/train/tensor_act_model_layers_25/mean":-0.0121307373046875,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/norm":0.028569278146505703,"train/train/tensor_param_model_layers_82_mlp_gate_proj_weight/max_abs":0.07763671875,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_5_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_mlp_down_proj/max_abs":0.08251953125,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/mean":-8.535385131835938e-05,"train/train/tensor_act_model_layers_3_input_layernorm/max_abs":5.03125,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn_o_proj/norm":277.1543360672444,"train/train/tensor_param_model_layers_90_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_32_post_attention_layernorm/max_abs":5.03125,"train/train/tensor_act_model_layers_84_post_attention_layernorm/mean":0.010528564453125,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/max_abs":0.000514984130859375,"train/train/tensor_act_model_layers_25_post_attention_layernorm/std":1.0000012274824,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_57/grad/std":0.0001932502415845275,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/std":0.00010378047838016346,"train/train/tensor_param_model_layers_34_mlp_up_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/max_abs":0.00110626220703125,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_36_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_30_self_attn_k_proj/max_abs":1.09375,"train/train/layer_model_layers_61/act/std":0.4238526077285235,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/max_abs":0.0013580322265625,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_q_proj/mean":0.00904083251953125,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/std":0.0196533203125,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_68/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_mlp_gate_proj_weight/mean":-5.1021575927734375e-05,"train/train/tensor_act_model_layers_55_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/std":0.0004758122375117473,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/std":8.455731022427346e-05,"train/train/tensor_act_model_layers_20/mean":-0.0132904052734375,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/max_abs":3.933906555175781e-06,"train/train/tensor_act_model_layers_28_self_attn_q_proj/mean":0.0024061203002929688,"train/train/tensor_grad_model_layers_85_mlp_gate_proj_weight/mean":-6.402842700481415e-07,"train/train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_post_attention_layernorm/max_abs":4.65625,"train/train/layer_model_layers_49/grad/max_abs":0.00543212890625,"train/train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/mean":-7.80448317527771e-07,"train/train/tensor_act_model_layers_18_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_39/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_self_attn/std":0.050539023130170065,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/mean":0.00014495849609375,"train/train/tensor_param_model_layers_51_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_12_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_/std":0,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/std":0.00039102638290545636,"train/train/tensor_act_model_layers_45_mlp/norm":91.43993236184221,"train/train/tensor_act_model_layers_77_self_attn_k_proj/norm":1310.872760853254,"train/train/layer_model_layers_88/grad/max_abs":0.0035858154296875,"train/train/tensor_act_model_layers_69_mlp/max_abs":0.08349609375,"train/train/tensor_act_model_layers_71_mlp/max_abs":0.08935546875,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/std":7.930101322446683e-05,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/max_abs":3.904104232788086e-06,"train/train/tensor_act_model_layers_34_self_attn_v_proj/norm":1323.0012659371596,"train/train/layer__model_layers_52/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_40/max_abs":1.5859375,"train/train/tensor_param_model_layers_85_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/mean":0.00015735626220703125,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/mean":-0.00015354156494140625,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/std":3.8316911870241143e-07,"train/train/tensor_act_model_layers_65_self_attn_o_proj/std":0.051026963392844904,"train/train/tensor_grad_model_layers_46_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn/mean":0.0007867813110351562,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/max_abs":0.07861328125,"train/train/tensor_act_model_layers_25_mlp_gate_proj/std":0.22607585525686355,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/mean":0.0002765655517578125,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/std":0.0198974609375,"train/train/layer_model_layers_18/act/norm":8976.359861788544,"train/train/tensor_act_model_layers_49_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_68_mlp/mean":-9.28640365600586e-05,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/max_abs":0.09033203125,"train/train/tensor_param_model_layers_85_mlp_gate_proj_weight/mean":0.0001468658447265625,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/std":5.747005221894461e-07,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/mean":-0.00018215179443359375,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/mean":4.336470738053322e-09,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/max_abs":0.005462646484375,"train/train/tensor_act_model_layers_90_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_mlp_up_proj/max_abs":1.078125,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/norm":0.00017346586071755412,"train/train/tensor_act_model_layers_88_self_attn/std":0.04809767991862529,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/global/act/mean":-0.0026260052889876483,"train/train/tensor_act_model_layers_0_post_attention_layernorm/max_abs":5.25,"train/train/tensor_param_model_layers_63_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn/max_abs":0.2421875,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_59_self_attn_q_proj/std":0.2260802366390253,"train/train/tensor_act_model_layers_33_input_layernorm/norm":5792.571044923815,"train/train/tensor_act_model_embed_tokens/max_abs":0.09912109375,"train/train/tensor_param_model_layers_76_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/max_abs":0.080078125,"train/train/layer_model_layers_16/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/mean":6.996095180511475e-06,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/norm":0.0007052316787384945,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/mean":-9.393692016601562e-05,"train/train/tensor_act_model_layers_87_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_28/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_post_attention_layernorm/std":1.0000010737270406,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/max_abs":0.000827789306640625,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/std":3.751367359085753e-07,"train/train/layer__model_layers_20/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp_down_proj/norm":89.93132066212658,"train/train/layer_model_layers_20/grad/frac_near_user_limit":0,"train/train/layer_model_layers_29/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/mean":0.0015850067138671875,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/norm":0.0023458372200046637,"train/train/tensor_param_model_layers_72_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/max_abs":0.08544921875,"train/train/tensor_param_model_layers_85_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/std":8.595516309729e-05,"train/train/tensor_param_model_layers_52_self_attn_k_proj_weight/mean":0.00013065338134765625,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_51_mlp_up_proj/norm":1832.2404298437873,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/max_abs":0.09375,"train/train/tensor_act_model_layers_86/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/norm":0.024780649266892916,"train/train/tensor_param_model_layers_16_self_attn_q_proj_weight/norm":2.546875,"train/train/layer_model_layers_84/act/std":0.42937447033019405,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_64_self_attn_q_proj/max_abs":1.125,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/max_abs":0.000354766845703125,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/mean":-2.1988525986671448e-06,"train/train/layer_model_layers_59/act/std":0.4237498670723525,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/std":7.869655401002824e-05,"train/train/tensor_act_model_layers_44_self_attn_k_proj/mean":0.017181396484375,"train/train/layer_model_layers_90/grad/norm":0.14256562799910216,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/mean":1.1436641216278076e-06,"train/train/layer_model_layers_54/grad/norm":0.16319371776027305,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/norm":0.0384972686046387,"train/train/tensor_act_model_layers_28_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_q_proj/max_abs":1.0234375,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/norm":0.00018665357964278283,"train/train/tensor_grad_model_layers_67_mlp_down_proj_weight/std":8.359121668919519e-05,"train/train/tensor_grad_model_layers_78_mlp_gate_proj_weight/mean":-1.0617077350616455e-06,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_up_proj/std":0.22607755246598427,"train/train/tensor_act_model_layers_49_input_layernorm/std":1.0000053088156218,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/max_abs":0.08251953125,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/mean":9.655195754021406e-09,"train/train/tensor_act_model_layers_19_self_attn_o_proj/std":0.049502737274082956,"train/train/layer_model_layers_72/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35/mean":-0.00458526611328125,"train/train/tensor_act_model_layers_14_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/mean":9.005889296531677e-07,"train/train/tensor_act_model_layers_70_mlp/max_abs":0.0869140625,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/std":4.716543714993653e-05,"train/train/tensor_act_model_layers_22_self_attn_o_proj/std":0.04705999858761875,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_mlp_up_proj_weight/max_abs":0.00152587890625,"train/train/tensor_param_model_layers_25_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13/max_abs":0.85546875,"train/train/tensor_grad_model_layers_60_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp_up_proj/mean":-0.003002166748046875,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23/norm":1377.9872432084692,"train/train/tensor_act_model_layers_66_self_attn_k_proj/mean":0.004344940185546875,"train/train/tensor_grad_model_layers_91_self_attn_o_proj_weight/std":0.0003400073812972497,"train/train/tensor_act_model_layers_20_self_attn_v_proj/std":0.2346233065867448,"train/train/tensor_act_model_layers_50_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/max_abs":7.420778274536133e-06,"train/train/tensor_act_model_layers_30_self_attn_k_proj/mean":0.016357421875,"train/train/layer_model_layers_18/act/max_abs":4.5625,"train/train/tensor_act_model_layers_21_self_attn_k_proj/std":0.22631998174717524,"train/train/tensor_grad_model_layers_88_self_attn_v_proj_weight/std":0.00035101084995179364,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/norm":9.341476032414227e-05,"train/train/tensor_act_model_layers_19_self_attn_v_proj/mean":0.011810302734375,"train/train/tensor_grad_model_layers_89_self_attn_k_proj_weight/mean":2.9940565582364798e-09,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/mean":-3.933906555175781e-05,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/mean":-1.621246337890625e-05,"train/train/tensor_act_model_layers_32_mlp_up_proj/std":0.22363550731646953,"train/train/tensor_act_model_layers_4_self_attn_v_proj/mean":-0.0109405517578125,"train/train/layer_model_layers_35/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn_v_proj/mean":0.0025882720947265625,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/std":0.00012125323709539692,"train/train/layer__model_layers_53/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp_up_proj/max_abs":1.1328125,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/max_abs":0.076171875,"train/train/tensor_act_model_layers_85_self_attn_q_proj/mean":0.00533294677734375,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/mean":0.00010013580322265625,"train/train/layer_model_layers_22/act/mean":-0.006548702716827393,"train/train/tensor_act_model_layers_88_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_post_attention_layernorm/std":1.0000009443606446,"train/train/tensor_grad_model_layers_61_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_27/grad/std":0.00029475360414081076,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/norm":0.16655544702652056,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/std":0.00018868494966156079,"train/train/tensor_param_model_layers_15_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_68_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/mean":-8.083879947662354e-06,"train/train/tensor_act_model_layers_1_mlp/std":0.01556400094021893,"train/train/tensor_param_model_layers_33_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_1_input_layernorm_weight/mean":7.05718994140625e-05,"train/train/tensor_act_model_layers_11_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_79/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/std":0.00010085907368503254,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/std":4.134120040644756e-07,"train/train/tensor_act_model_layers_29_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_44/param/norm":17.95223425771149,"train/train/tensor_grad_model_layers_85_mlp_gate_proj_weight/max_abs":0.00131988525390625,"train/train/tensor_act_model_layers_29_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_71/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_30_post_attention_layernorm/max_abs":4.84375,"train/train/tensor_param_model_layers_3_mlp_gate_proj_weight/norm":3.59375,"train/train/tensor_param_model_layers_28_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/mean":1.9476283341646194e-07,"train/train/tensor_act_model_layers_26/std":0.2492739123661416,"train/train/tensor_grad_model_layers_53_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/max_abs":0.00628662109375,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/mean":2.50060111284256e-07,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/std":5.305863059593774e-05,"train/train/tensor_act_model_layers_18_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/max_abs":0.0810546875,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/std":5.363369832803224e-07,"train/train/tensor_act_model_layers_81_mlp_up_proj/norm":1836.4129069380162,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/norm":2.53125,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/max_abs":0.08056640625,"train/train/tensor_act_model_layers_18_self_attn_q_proj/mean":-0.016510009765625,"train/train/tensor_param_model_layers_49_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_50_self_attn_o_proj/mean":-0.0005951225757598877,"train/train/tensor_act_model_layers_92/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_mlp_gate_proj_weight/max_abs":0.00124359130859375,"train/train/tensor_act_model_layers_12_post_attention_layernorm/norm":5792.5000000002365,"train/train/tensor_act_model_layers_61/max_abs":1.828125,"train/train/tensor_act_model_layers_53_post_attention_layernorm/max_abs":4.9375,"train/train/layer_model_layers_37/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn_k_proj/norm":1274.4910267576624,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/std":3.509245424278778e-05,"train/train/tensor_act_model_layers_39_mlp/max_abs":0.08251953125,"train/train/tensor_act_model_layers_20_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_self_attn_o_proj/std":0.047242308874641745,"train/train/tensor_act_model_layers_81_mlp_down_proj/max_abs":0.080078125,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_69_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn_o_proj/max_abs":0.23046875,"train/train/tensor_grad_model_layers_46_mlp_gate_proj_weight/mean":1.6472768038511276e-07,"train/train/tensor_param_model_layers_60_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/max_abs":5.5730342864990234e-06,"train/train/tensor_param_model_layers_69_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_91_self_attn_v_proj/norm":1262.4792065908807,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/std":0.001923702083658243,"train/train/tensor_act_model_layers_28_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_35/act/mean":-0.003295966557094029,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/norm":0.2447828467112583,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/mean":1.2731179594993591e-06,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29/std":0.2656287918801893,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_1_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/mean":0.00022220611572265625,"train/train/tensor_act_model_layers_85_mlp_up_proj/norm":1869.8630746129465,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/max_abs":0.08056640625,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/std":0.00043628575836246746,"train/train/layer_model_layers_62/grad/norm":0.17378073254373225,"train/train/tensor_act_model_layers_88_mlp/std":0.01550340960988413,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/max_abs":0.08203125,"train/train/tensor_grad_model_layers_65_mlp_gate_proj_weight/mean":4.880130290985107e-07,"train/train/tensor_grad_model_layers_72_self_attn_o_proj_weight/max_abs":0.00506591796875,"train/train/layer__model_layers_52/param/norm":17.940466871363743,"train/train/tensor_act_model_layers_24_mlp/max_abs":0.083984375,"train/train/tensor_act_model_layers_10_self_attn_k_proj/max_abs":1.0859375,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/norm":0.0003240163752103529,"train/train/tensor_act_model_layers_81_self_attn/max_abs":0.23046875,"train/train/tensor_act_model_layers_56_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/mean":-7.785274647176266e-10,"train/train/tensor_act_model_layers_76_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_post_attention_layernorm_weight/max_abs":0.0002193450927734375,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/mean":6.400048732757568e-06,"train/train/tensor_act_model_layers_80_mlp/max_abs":0.0869140625,"train/train/tensor_act_model_layers_1_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/mean":6.062909960746765e-07,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/max_abs":9.918212890625e-05,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/max_abs":0.08154296875,"train/train/tensor_act_model_layers_54_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/mean":7.898779585957527e-08,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/std":3.957461617731884e-07,"train/train/layer_model_layers_56/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_k_proj_weight/max_abs":4.947185516357422e-06,"train/train/tensor_act_model_layers_62_mlp_gate_proj/mean":-0.0135345458984375,"train/train/tensor_grad_model_layers_4_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_self_attn/mean":0.0011844635009765625,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/norm":0.28586771796208743,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_82_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp/max_abs":0.080078125,"train/train/tensor_act_model_layers_23_input_layernorm/max_abs":4.78125,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/mean":-1.0088086128234863e-05,"train/train/tensor_act_model_layers_79_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/norm":2.53125,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/std":0.020263671875,"train/train/tensor_grad_model_layers_51_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/norm":0.0022984872677106844,"train/train/layer_model_layers_63/grad/norm":0.16435802113174755,"train/train/tensor_act_model_layers_70_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/max_abs":0.005096435546875,"train/train/tensor_param_model_layers_77_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/std":0.00040238191161602556,"train/train/layer__model_layers_55/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/norm":0.0027662316084061615,"train/train/tensor_act_model_layers_34_mlp_up_proj/std":0.2251049585732749,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/max_abs":6.735324859619141e-06,"train/train/layer_model_layers_50/grad/norm":0.17352386649273435,"train/train/layer_model_layers_13/act/frac_near_dtype_limit":0,"train/train/global/act/norm":89131.68564614924,"train/train/tensor_act_model_layers_13_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_56/grad/std":0.00020960538498012164,"train/train/tensor_grad_model_layers_5_self_attn_o_proj_weight/mean":2.137385308742523e-06,"train/train/tensor_act_model_layers_26_self_attn_k_proj/norm":1278.4512688568907,"train/train/tensor_act_model_layers_70_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_mlp_gate_proj_weight/norm":0.061874116833481665,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/mean":6.211921572685242e-07,"train/train/tensor_act_model_layers_39_self_attn_q_proj/mean":-0.0029125213623046875,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/norm":0.1275863115962613,"train/train/tensor_act_model_layers_53_mlp/std":0.015198404618691739,"train/train/tensor_act_model_layers_11_post_attention_layernorm/max_abs":4.78125,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/std":4.2959492191193386e-07,"train/train/tensor_act_model_layers_13_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_13_self_attn_o_proj/max_abs":0.251953125,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp/mean":0.00014853477478027344,"train/train/tensor_act_model_layers_57_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_21/act/max_abs":4.5,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/norm":0.00012526758094989787,"train/train/tensor_param_model_layers_63_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/max_abs":0.08154296875,"train/train/tensor_act_model_layers_50_mlp_up_proj/max_abs":1.09375,"train/train/layer_model_layers_17/grad/std":0.0004217698880484146,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/mean":5.082256393507123e-09,"train/train/tensor_param_model_layers_81_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/mean":1.5720725059509277e-05,"train/train/tensor_act_model_layers_65_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_up_proj/std":0.22193080064418724,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/norm":0.0008155694287878993,"train/train/tensor_act_model_layers_92_mlp_up_proj/std":0.22802942334132345,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/max_abs":9.119510650634766e-06,"train/train/tensor_act_model_layers_20_mlp_up_proj/std":0.22680819688474543,"train/train/tensor_act_model_layers_80/mean":0.00516510009765625,"train/train/tensor_act_model_layers_39_self_attn_k_proj/max_abs":1.15625,"train/train/tensor_act_model_layers_47_input_layernorm/std":1.0000055002814119,"train/train/tensor_act_model_layers_56_mlp/max_abs":0.08251953125,"train/train/tensor_act_model_layers_90_self_attn_q_proj/max_abs":1.1328125,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_84_input_layernorm/std":1.0000006210464452,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/std":0.00046687833905538896,"train/train/tensor_act_model_layers_84_mlp_gate_proj/norm":1838.4320311920253,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/std":0.00011204342538218878,"train/train/tensor_act_model_layers_63/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/max_abs":0.0771484375,"train/train/tensor_act_model_layers_67_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/max_abs":0.0849609375,"train/train/tensor_act_model_layers_51_self_attn_v_proj/std":0.2292539557202269,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/mean":0.00017070770263671875,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/mean":1.5858560800552368e-05,"train/train/tensor_act_model_layers_82_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/std":9.480535773686915e-05,"train/train/layer__model_layers_11/param/max_abs":1,"train/train/tensor_act_model_layers_22_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/mean":-0.0003910064697265625,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/mean":-3.0850060284137726e-09,"train/train/tensor_act_model_layers_19_post_attention_layernorm/mean":-0.0560302734375,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/mean":-1.4379620552062988e-06,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/norm":0.04662143152607595,"train/train/layer_model_layers_1/grad/max_abs":0.03173828125,"train/train/tensor_act_model_layers_56_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/mean":-1.5720615920145065e-09,"train/train/tensor_act_model_layers_41_self_attn_v_proj/mean":0.012176513671875,"train/train/tensor_act_model_layers_63_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_k_proj_weight/norm":9.29193399362425e-05,"train/train/tensor_act_model_layers_88_self_attn_q_proj/max_abs":1.0625,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_input_layernorm/norm":5792.5942382892035,"train/train/tensor_act_model_layers_49_mlp_down_proj/mean":0.00016617774963378906,"train/train/tensor_act_model_layers_39/mean":-0.005340576171875,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/mean":0.0002193450927734375,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64/norm":2310.7722633815097,"train/train/tensor_param_model_layers_41_mlp_up_proj_weight/mean":2.8252601623535156e-05,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/std":0.000904467408785867,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/std":0.0005901282058969872,"train/train/tensor_act_model_layers_74_mlp_gate_proj/max_abs":1.1015625,"train/train/layer_model_layers_77/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_42_mlp_gate_proj/mean":-0.0006856918334960938,"train/train/tensor_act_model_layers_82_mlp_up_proj/max_abs":1.171875,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/max_abs":0.005279541015625,"train/train/tensor_act_model_layers_48_input_layernorm/std":1.0000050545949801,"train/train/layer_model_layers_32/act/std":0.41678901613288616,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/max_abs":0.08447265625,"train/train/tensor_act_model_layers_58_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_24/act/max_abs":4.75,"train/train/tensor_param_model_layers_88_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_v_proj/std":0.23535465795435603,"train/train/tensor_param_model_layers_34_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/std":0.0004072370804353662,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_input_layernorm_weight/std":0,"train/train/layer__model_layers_50/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_24/act/frac_near_user_limit":0,"train/train/layer_model_layers_91/grad/max_abs":0.00396728515625,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_input_layernorm/mean":-0.02490234375,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/std":7.291207129302815e-07,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_48_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_up_proj/max_abs":1.03125,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_40_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/mean":4.1620805859565735e-06,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/mean":-7.295608520507812e-05,"train/train/tensor_act_model_layers_27_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/mean":-0.0004215240478515625,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp_gate_proj/mean":-0.0023581981658935547,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_52_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_27/grad/max_abs":0.0072021484375,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/std":7.274700015668998e-05,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/max_abs":0.08984375,"train/train/tensor_act_model_layers_25_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_k_proj_weight/std":4.1440864397693415e-07,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/max_abs":4.0531158447265625e-06,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/norm":0.15603164674634742,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/mean":-0.000102996826171875,"train/train/tensor_act_model_layers_32_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/max_abs":0.00107574462890625,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/mean":3.691762685775757e-06,"train/train/tensor_act_model_layers_29_mlp_down_proj/std":0.015367045917187053,"train/train/tensor_act_model_layers_47_self_attn_k_proj/mean":0.0078125,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn_q_proj/norm":1326.8819437937136,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_k_proj/norm":1259.1991839705945,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/mean":7.390975952148438e-05,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/max_abs":0.07666015625,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/mean":3.574677975848317e-08,"train/train/layer_model_layers_37/act/max_abs":5.03125,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/max_abs":0.07421875,"train/train/tensor_param_model_layers_86_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_48_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2/norm":289.10550169277724,"train/train/layer_model_layers_5/grad/max_abs":0.01434326171875,"train/train/layer_model_layers_57/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/std":8.787662928765872e-05,"train/train/tensor_act_model_layers_87_self_attn_o_proj/mean":-0.0015964508056640625,"train/train/tensor_act_model_layers_83_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/max_abs":0.00151824951171875,"train/train/tensor_act_model_layers_87_mlp/max_abs":0.08056640625,"train/train/layer_model_layers_17/act/std":0.4143178563346527,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/mean":1.3817043509334326e-08,"train/train/tensor_param_model_layers_86_mlp_up_proj_weight/mean":0.000125885009765625,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/mean":-2.0951032638549805e-05,"train/train/tensor_act_model_layers_68_mlp_down_proj/max_abs":0.0791015625,"train/train/tensor_act_model_layers_52_mlp_gate_proj/max_abs":1.1171875,"train/train/tensor_act_model_layers_74_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/norm":0.0003999717359486202,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/mean":0.000133514404296875,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/mean":2.046581357717514e-07,"train/train/tensor_act_model_layers_1/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/max_abs":5.9604644775390625e-06,"train/train/tensor_param_model_layers_68_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_66_self_attn_v_proj/max_abs":1.1484375,"train/train/tensor_act_model_layers_2_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/max_abs":0.07568359375,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/max_abs":0.08056640625,"train/train/tensor_act_model_layers_64_self_attn_o_proj/norm":293.0439414109723,"train/train/tensor_act_model_layers_83_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_post_attention_layernorm_weight/mean":-2.8431415557861328e-05,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84/std":0.4624117829020018,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/std":0.00014630983454683668,"train/train/tensor_param_model_layers_66_self_attn_q_proj_weight/max_abs":0.07568359375,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/std":0.000443360457294285,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/mean":4.0531158447265625e-05,"train/train/tensor_grad_model_layers_44_input_layernorm_weight/mean":5.615875124931335e-07,"train/train/tensor_act_model_layers_36/frac_near_user_limit":0,"train/train/layer__model_layers_45/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_post_attention_layernorm/norm":5792.593627933251,"train/train/tensor_param_model_layers_51_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_31_post_attention_layernorm/max_abs":4.90625,"train/train/tensor_act_model_layers_4/frac_near_user_limit":0,"train/train/tensor_param_model_layers_88_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_42_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/mean":-0.00013828277587890625,"train/train/layer__model_layers_59/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp_up_proj/max_abs":1.2109375,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/max_abs":1.1324882507324219e-05,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_up_proj/max_abs":1.140625,"train/train/layer_model_layers_32/act/mean":-0.0050269535609654015,"train/train/tensor_act_model_layers_64_self_attn/max_abs":0.236328125,"train/train/tensor_grad_model_layers_7_mlp_gate_proj_weight/std":0.00025975377385153953,"train/train/layer_model_layers_4/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_v_proj/norm":1305.4188865045408,"train/train/tensor_act_model_layers_74_self_attn_k_proj/std":0.22388067003199336,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn/norm":287.18764228872,"train/train/tensor_param_model_layers_3_mlp_gate_proj_weight/mean":3.266334533691406e-05,"train/train/tensor_act_model_layers_90/max_abs":2.34375,"train/train/tensor_act_model_layers_69/std":0.4165167369199684,"train/train/tensor_grad_model_layers_22_self_attn_v_proj_weight/max_abs":0.00799560546875,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/mean":4.955567419528961e-06,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_75_input_layernorm/mean":0.0027484893798828125,"train/train/tensor_param_model_layers_66_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_5/param/std":0.04424328761137998,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/mean":-5.561742000281811e-08,"train/train/tensor_param_model_layers_5_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/std":9.169548753235166e-05,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/max_abs":0.0810546875,"train/train/tensor_param_model_layers_60_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_down_proj/max_abs":0.0869140625,"train/train/tensor_act_model_layers_80_mlp_gate_proj/norm":1838.7184005595673,"train/train/tensor_act_model_layers_13_mlp_down_proj/mean":-7.62939453125e-06,"train/train/tensor_act_model_layers_21_self_attn_q_proj/mean":0.0081024169921875,"train/train/layer__model_layers_55/param/mean":0.0015742291526377851,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_q_proj_weight/std":1.3365095395421579e-06,"train/train/tensor_act_model_layers_74_input_layernorm/mean":0.00171661376953125,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/mean":-6.437301635742188e-05,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/norm":0.0019461574550620614,"train/train/tensor_param_model_layers_71_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/std":0.0004804083375826175,"train/train/tensor_param_model_layers_4_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/norm":0.0003798234702919574,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_46_mlp_gate_proj_weight/norm":0.03061611289873506,"train/train/tensor_param_model_layers_92_mlp_gate_proj_weight/mean":0.00017070770263671875,"train/train/layer_model_layers_71/act/std":0.4273032466411482,"train/train/tensor_act_model_layers_62_self_attn_o_proj/max_abs":0.240234375,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/max_abs":3.0517578125e-05,"train/train/tensor_act_model_layers_7_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/mean":1.7535057850182056e-08,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/mean":-9.584426879882812e-05,"train/train/tensor_act_model_layers_76_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76/norm":2538.7593648213447,"train/train/tensor_param_model_layers_42_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_46/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/mean":2.6345252990722656e-05,"train/train/tensor_act_model_layers_83_self_attn_o_proj/norm":286.67480820701564,"train/train/tensor_param_model_layers_82_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_input_layernorm/max_abs":4.6875,"train/train/tensor_act_model_layers_77_self_attn_q_proj/norm":1268.9417312396065,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_11/act/std":0.41309550587677796,"train/train/layer_model_layers_9/act/max_abs":5,"train/train/tensor_act_model_layers_75_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_mlp_gate_proj_weight/norm":0.030773947570865698,"train/train/tensor_param_model_layers_44_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/max_abs":0.08935546875,"train/train/tensor_act_model_layers_22_mlp/std":0.01586961476140366,"train/train/global/grad/mean":-7.643503142972937e-08,"train/train/tensor_act_model_layers_20_input_layernorm/std":0.9990349747484736,"train/train/tensor_act_model_layers_60_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_input_layernorm/max_abs":4.5625,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_mlp_down_proj/mean":0.00018405914306640625,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/std":0.00028063639476304394,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/max_abs":0.004486083984375,"train/train/layer_model_layers_78/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp_gate_proj/std":0.22729842221844215,"train/train/layer__model_layers_57/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_89_input_layernorm/std":1.000004337509069,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/max_abs":0.00061798095703125,"train/train/tensor_param_model_layers_85_self_attn_q_proj_weight/max_abs":0.09033203125,"train/train/tensor_act_model_layers_0_mlp_gate_proj/std":0.22534225110972242,"train/train/tensor_act_model_layers_21/mean":-0.013336181640625,"train/train/tensor_act_model_layers_20_self_attn_v_proj/mean":0.016693115234375,"train/train/tensor_act_model_layers_49_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp/max_abs":0.0830078125,"train/train/tensor_param_model_layers_67_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_52/grad/std":0.00021633094702836557,"train/train/tensor_act_model_layers_32_self_attn_q_proj/std":0.22388033823860848,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/max_abs":0.006500244140625,"train/train/layer__model_layers_29/param/norm":17.919020860352553,"train/train/layer_model_layers_7/act/mean":-0.01568491118294852,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp_gate_proj/mean":0.004604339599609375,"train/train/layer_model_layers_61/act/mean":0.00012983168874468123,"train/train/tensor_act_model_layers_7_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_17/param/mean":0.0015280927399205343,"train/train/tensor_act_model_layers_80_self_attn_v_proj/mean":0.01068115234375,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/std":7.633335445359652e-05,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/std":7.225079954918349e-05,"train/train/tensor_act_model_layers_50_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer_model_layers_74/grad/max_abs":0.005157470703125,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/std":0.00011699082692746425,"train/train/tensor_act_model_layers_93_self_attn/norm":292.288934025549,"train/train/tensor_act_model_layers_10_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_11_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_76_post_attention_layernorm/mean":0.003299713134765625,"train/train/tensor_act_model_layers_67_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_19/param/norm":17.943106702157237,"train/train/tensor_param_model_layers_76_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/mean":-9.34302806854248e-06,"train/train/tensor_act_model_layers_52_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/max_abs":0.07763671875,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_89_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_7/act/max_abs":5.21875,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_78/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/std":3.634132965460169e-07,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/mean":-4.023313522338867e-06,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/std":0.00015353200797587387,"train/train/layer_model_layers_58/grad/norm":0.16830615771235205,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_13_mlp/max_abs":0.08056640625,"train/train/tensor_act_model_layers_88_mlp_up_proj/std":0.22363484539306341,"train/train/tensor_act_model_layers_71_self_attn_q_proj/max_abs":1.0703125,"train/train/tensor_act_model_layers_66_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_16/grad/std":0.0004544104655195801,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/mean":4.62478055851534e-10,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/mean":-2.0245352061465383e-09,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/mean":4.495632310863584e-09,"train/train/tensor_act_model_layers_53_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_k_proj_weight/norm":0.00022456802776059316,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/norm":0.00013597469541183705,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/norm":0.00010948060696301413,"train/train/tensor_act_model_layers_47/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_post_attention_layernorm/norm":5792.554809584804,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/std":3.502299762361152e-05,"train/train/tensor_act_model_layers_65_self_attn_k_proj/max_abs":1.0703125,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/std":1.5623997970368026e-06,"train/train/layer__model_layers_15/param/max_abs":1,"train/train/tensor_grad_model_embed_tokens_weight/mean":4.4051557779312134e-07,"train/train/tensor_act_model_layers_59_self_attn_o_proj/std":0.0496236921594323,"train/train/tensor_param_model_layers_64_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/std":4.519260820754232e-07,"train/train/tensor_param_model_layers_3_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_v_proj/max_abs":1.03125,"train/train/tensor_act_model_layers_71/mean":0.00016617774963378906,"train/train/layer_model_layers_38/act/max_abs":5,"train/train/tensor_act_model_layers_90_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/max_abs":0.091796875,"train/train/tensor_act_model_layers_60_post_attention_layernorm/max_abs":4.625,"train/train/tensor_param_model_layers_20_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/norm":0.028873306266023604,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_24_self_attn_q_proj/max_abs":1.078125,"train/train/tensor_act_model_layers_4_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/norm":0.04148828549102655,"train/train/tensor_param_model_layers_26_mlp_gate_proj_weight/max_abs":0.07666015625,"train/train/tensor_act_model_layers_66_mlp/max_abs":0.076171875,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/max_abs":0.004058837890625,"train/train/tensor_param_model_layers_17_mlp_up_proj_weight/norm":3.65625,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_q_proj/max_abs":1.015625,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/mean":-9.210780262947083e-07,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_22/max_abs":1.1640625,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/max_abs":0.0093994140625,"train/train/tensor_act_model_layers_2_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_mlp_gate_proj_weight/max_abs":0.08349609375,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_gate_proj_weight/norm":0.09664423128644815,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_65_self_attn_q_proj/std":0.22632454994319115,"train/train/layer_model_layers_13/act/std":0.4137501713628655,"train/train/tensor_grad_model_layers_62_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_78_mlp_gate_proj_weight/norm":0.029867340271100305,"train/train/tensor_act_model_layers_9_self_attn/mean":-0.0008592605590820312,"train/train/tensor_act_model_layers_34_input_layernorm/max_abs":5.0625,"train/train/tensor_act_model_layers_72/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp/std":0.015686609768660917,"train/train/layer__model_layers_66/param/frac_near_user_limit":0,"train/train/layer_model_layers_18/grad/norm":0.31720148870457276,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/std":8.615965076604688e-05,"train/train/tensor_act_model_layers_68_mlp_up_proj/frac_near_user_limit":0,"train/train/layer__model_layers_7/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/mean":2.8191134333610535e-06,"train/train/tensor_act_model_layers_74_mlp_gate_proj/std":0.22705541459862696,"train/train/tensor_act_model_layers_63_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_7/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_down_proj/std":0.015503836343847308,"train/train/tensor_act_model_layers_43_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/mean":-5.615875124931335e-07,"train/train/global/param/frac_near_user_limit":0,"train/train/layer_model_layers_2/act/mean":-0.003129448209490095,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/norm":0.10386110630144782,"train/train/tensor_act_model_layers_43/std":0.3193385864096546,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/max_abs":0.004241943359375,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_mlp_up_proj/std":0.23022746127840948,"train/train/tensor_grad_model_layers_72_mlp_up_proj_weight/std":7.130668958420769e-05,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/norm":0.11168893269461573,"train/train/tensor_grad_model_layers_85_mlp_gate_proj_weight/norm":0.026944239160623086,"train/train/tensor_grad_model_layers_65_mlp_gate_proj_weight/std":8.502661127426838e-05,"train/train/layer_model_layers_22/act/max_abs":4.6875,"train/train/tensor_act_model_layers_71_self_attn/max_abs":0.2890625,"train/train/tensor_param_model_layers_70_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_8_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_mlp_up_proj/mean":0.0120849609375,"train/train/tensor_act_model_layers_74_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_q_proj/std":0.2204632123367993,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/mean":-7.795169949531555e-07,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/mean":9.918212890625e-05,"train/train/layer__model_layers_40/param/mean":0.0015333461315137176,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/norm":0.1461796961638539,"train/train/tensor_act_model_layers_42_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_o_proj_weight/mean":-5.5693089962005615e-06,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_16_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_78_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_post_attention_layernorm/norm":5792.583251953727,"train/train/tensor_act_model_layers_22_self_attn_q_proj/std":0.22120542184475503,"train/train/tensor_act_model_layers_5_mlp_up_proj/norm":1868.3744504186943,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/max_abs":0.08056640625,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_v_proj/std":0.2277889316151293,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/max_abs":0.09033203125,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_32_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_o_proj/mean":6.67572021484375e-05,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_65/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp_up_proj/frac_near_user_limit":0,"train/train/layer__model_layers_58/param/mean":0.001564811432789343,"train/train/layer__model_layers_29/param/std":0.04421175276498373,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/max_abs":0.0009613037109375,"train/train/tensor_act_model_layers_4_self_attn_k_proj/mean":0.01727294921875,"train/train/tensor_grad_model_layers_0_mlp_gate_proj_weight/norm":0.3111190802446631,"train/train/tensor_act_model_layers_55_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/std":3.5512832427816254e-07,"train/train/tensor_act_model_layers_79/max_abs":2.109375,"train/train/layer_model_layers_39/grad/mean":3.5742636662493815e-07,"train/train/layer__model_layers_68/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp_up_proj/max_abs":1.109375,"train/train/tensor_act_model_layers_16_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/std":6.499564205164966e-05,"train/train/tensor_act_model_layers_74_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/norm":0.00079837294095812,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_83/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_72_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_self_attn_v_proj/max_abs":1.1015625,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/std":0.00042842174036887646,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/std":0.015244060172788388,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/std":0.00011690239317207447,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/norm":0.11679264941274406,"train/train/tensor_act_model_layers_18_self_attn_v_proj/std":0.21460275137796028,"train/train/tensor_act_model_layers_48_self_attn_q_proj/norm":1271.7907631144835,"train/train/tensor_act_model_layers_0_self_attn_v_proj/norm":1296.2482577728445,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55/std":0.36671058513768395,"train/train/tensor_param_model_layers_92_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_8/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_20/act/std":0.4163855424529056,"train/train/tensor_act_model_layers_33_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_o_proj_weight/mean":1.1457595974206924e-06,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/mean":-2.532266080379486e-06,"train/train/tensor_act_model_layers_13_mlp/norm":91.64267406011611,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_up_proj/max_abs":1.0703125,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_k_proj_weight/max_abs":1.3768672943115234e-05,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/std":0.00011166502153665844,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_mlp_gate_proj/max_abs":1.1015625,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/max_abs":0.006103515625,"train/train/layer__model_layers_25/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_83/act/max_abs":4.71875,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_self_attn_v_proj/max_abs":1.1796875,"train/train/tensor_grad_model_layers_76_mlp_gate_proj_weight/std":7.169031598472019e-05,"train/train/tensor_param_model_layers_91_input_layernorm_weight/std":0,"train/train/layer_model_layers_52/act/norm":9133.722359958894,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/mean":-1.3709068298339844e-05,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/mean":-0.00015354156494140625,"train/train/tensor_act_model_layers_79_self_attn_k_proj/std":0.22877282707209842,"train/train/layer_model_layers_63/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45/max_abs":1.6328125,"train/train/tensor_act_model_layers_14_mlp_gate_proj/mean":0.0022106170654296875,"train/train/tensor_param_model_layers_86_mlp_down_proj_weight/max_abs":0.0849609375,"train/train/layer_model_layers_47/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_q_proj/max_abs":1.03125,"train/train/tensor_act_model_layers_2/std":0.04992791914913544,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/max_abs":0.000835418701171875,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_56/param/mean":0.0015267418252882458,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_18_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/norm":2.5625,"train/train/layer__model_layers_26/param/max_abs":1,"train/train/tensor_grad_model_layers_45_mlp_gate_proj_weight/max_abs":0.00125885009765625,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/norm":0.1061082455335241,"train/train/tensor_act_model_layers_93_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/mean":-5.252659320831299e-07,"train/train/tensor_act_model_layers_48/max_abs":1.7265625,"train/train/layer_model_layers_82/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_v_proj/std":0.22558996672769743,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/std":0.00019284527004614043,"train/train/tensor_param_model_layers_17_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn_v_proj/std":0.22974558445751528,"train/train/tensor_param_model_layers_6_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_12_mlp/std":0.015381691709915752,"train/train/layer__model_layers_2/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_80_mlp_down_proj/norm":90.12947819630477,"train/train/tensor_act_model_layers_53_self_attn_k_proj/mean":-0.0087432861328125,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/std":0.0010885486092829916,"train/train/layer_model_layers_59/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/max_abs":0.07666015625,"train/train/tensor_act_model_layers_30_mlp/mean":-0.001148223876953125,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_input_layernorm/max_abs":4.875,"train/train/tensor_param_model_layers_5_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90/frac_near_user_limit":0,"train/train/layer_model_layers_42/act/norm":9093.62343487199,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_36_input_layernorm_weight/norm":0.00279220205400876,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/std":9.472973149623317e-05,"train/train/layer_model_layers_39/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer__model_layers_36/param/max_abs":1,"train/train/tensor_act_model_layers_16_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_post_attention_layernorm/std":1.0000022556607318,"train/train/layer__model_layers_74/param/std":0.044235029675742256,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_15_self_attn_v_proj/norm":1312.223426491269,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/max_abs":3.3527612686157227e-06,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/mean":3.695487976074219e-05,"train/train/tensor_act_model_layers_88_mlp_gate_proj/max_abs":1.0625,"train/train/tensor_param_model_layers_24_self_attn_k_proj_weight/norm":2.59375,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/mean":4.526227712631226e-06,"train/train/tensor_act_model_layers_73_self_attn_o_proj/norm":283.4490299270647,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/norm":0.1510723565152678,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/std":8.06045375431578e-05,"train/train/tensor_param_model_layers_12_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_4_self_attn_q_proj/std":0.23047118266486666,"train/train/tensor_param_model_layers_24_input_layernorm_weight/max_abs":1,"train/train/layer__model_layers_78/param/mean":0.0015700841656713144,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/std":6.632592237677898e-07,"train/train/tensor_act_model_layers_93_self_attn_k_proj/mean":-0.0099639892578125,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_26_self_attn_q_proj/norm":1292.6480260539108,"train/train/layer__model_layers_18/param/std":0.04425189527617165,"train/train/tensor_act_model_layers_49_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn/norm":287.5787673178776,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/mean":3.864988684654236e-06,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/max_abs":0.0023651123046875,"train/train/tensor_param_model_layers_65_self_attn_v_proj_weight/mean":-5.078315734863281e-05,"train/train/tensor_act_model_layers_9_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/mean":-2.384185791015625e-05,"train/train/tensor_act_model_layers_50_self_attn/std":0.047854814129641016,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/max_abs":0.07666015625,"train/train/tensor_act_model_layers_81_mlp_gate_proj/std":0.22559478100455738,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/mean":-3.337860107421875e-06,"train/train/tensor_act_model_layers_50_input_layernorm/std":1.0000033467238996,"train/train/tensor_grad_model_layers_58_mlp_up_proj_weight/std":7.797938448355493e-05,"train/train/tensor_param_model_layers_68_mlp_gate_proj_weight/mean":3.814697265625e-06,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/norm":0.15594911214827092,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/global_step":40,"train/train/tensor_param_model_layers_30_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_mlp_gate_proj_weight/max_abs":0.0791015625,"train/train/layer__model_layers_82/param/mean":0.0014412295241809673,"train/train/tensor_act_model_layers_37_mlp/std":0.01590006214206699,"train/train/tensor_grad_model_layers_15_mlp_gate_proj_weight/mean":1.1557713150978088e-06,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/mean":2.117827534675598e-06,"train/train/tensor_param_model_layers_63_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_20_self_attn_k_proj/max_abs":1.21875,"train/train/tensor_act_model_layers_84_self_attn/mean":0.00147247314453125,"train/train/tensor_act_model_layers_24_self_attn/std":0.04657170657770708,"train/train/tensor_act_model_layers_46_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn_k_proj/std":0.22290548545821426,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/max_abs":0.0712890625,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_25_mlp_gate_proj_weight/mean":-7.795169949531555e-07,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn/norm":289.9886559105947,"train/train/layer_model_layers_46/act/norm":9107.467565599622,"train/train/tensor_act_model_layers_3_self_attn/norm":226.96487138190938,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/mean":-3.0994415283203125e-05,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_self_attn/max_abs":0.2333984375,"train/train/tensor_act_model_layers_27_self_attn_v_proj/std":0.22046453459988588,"train/train/layer_model_layers_11/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_61/param/std":0.04425731581716744,"train/train/layer__model_layers_45/param/max_abs":1,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/max_abs":5.1875,"train/train/tensor_act_model_layers_63_self_attn_k_proj/norm":1314.7104229980898,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp_down_proj/mean":-0.0003509521484375,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/max_abs":0.08935546875,"train/train/tensor_act_model_layers_61_self_attn/mean":0.0026397705078125,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/norm":0.06073203538648078,"train/train/tensor_act_model_layers_23_mlp_down_proj/norm":89.8131948203714,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_9_self_attn_v_proj_weight/max_abs":0.00787353515625,"train/train/tensor_grad_model_layers_73_post_attention_layernorm_weight/std":4.7567595667108216e-05,"train/train/tensor_act_model_layers_91_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_o_proj/std":0.04657170657770708,"train/train/tensor_act_model_layers_92_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68/max_abs":1.953125,"train/train/layer_model_layers_5/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp_down_proj/max_abs":0.08935546875,"train/train/tensor_act_model_layers_59/mean":-0.0031070709228515625,"train/train/tensor_act_model_layers_2_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/max_abs":0.08203125,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_mlp/mean":0.0006856918334960938,"train/train/tensor_param_model_layers_90_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/std":3.13075553212354e-05,"train/train/tensor_param_model_layers_3_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_10_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_36/grad/mean":1.998468321804742e-07,"train/train/tensor_act_model_layers_57_post_attention_layernorm/std":1.0000088121935813,"train/train/tensor_act_model_layers_61_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer_model_layers_77/act/mean":-0.001500436237880162,"train/train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_mlp/std":0.016236471372754274,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/mean":-0.00015544891357421875,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_83/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_49/max_abs":1.7265625,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp_down_proj/max_abs":0.08740234375,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_14/act/frac_near_user_limit":0,"train/train/layer_model_layers_70/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_34/grad/mean":1.4269975843649016e-07,"train/train/tensor_param_model_layers_58_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/std":4.937729100545982e-07,"train/train/tensor_act_model_layers_17_mlp_down_proj/std":0.015564675963508922,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/std":0.00012079401763718882,"train/train/tensor_act_model_layers_86_self_attn/max_abs":0.2412109375,"train/train/tensor_param_model_layers_37_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_56_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_20_self_attn_o_proj/max_abs":0.25390625,"train/train/tensor_act_model_layers_46_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_51/grad/std":0.0002094731723441748,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/max_abs":0.078125,"train/train/tensor_grad_model_layers_50_mlp_gate_proj_weight/max_abs":0.002197265625,"train/train/tensor_act_model_layers_93_self_attn_o_proj/mean":0.0019741058349609375,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn/mean":-0.001697540283203125,"train/train/tensor_act_model_layers_28_post_attention_layernorm/std":1.0000030933831467,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_43_post_attention_layernorm/mean":-0.016815185546875,"train/train/tensor_act_model_layers_84/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_mlp_gate_proj/mean":0.002161264419555664,"train/train/tensor_param_model_layers_58_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp_up_proj/mean":-0.00737762451171875,"train/train/tensor_act_model_layers_37_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_mlp_gate_proj_weight/mean":-2.9624789021909237e-07,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_mlp_gate_proj_weight/max_abs":0.0849609375,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/max_abs":0.00213623046875,"train/train/tensor_grad_model_layers_37_mlp_gate_proj_weight/std":0.00010640076961856104,"train/train/tensor_act_model_layers_9_self_attn_o_proj/max_abs":0.2197265625,"train/train/tensor_param_model_layers_64_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_16_mlp_up_proj/norm":1866.8413202274685,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_82_self_attn/mean":0.0014495849609375,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/std":9.795987895212856e-05,"train/train/tensor_act_model_layers_26_mlp_gate_proj/std":0.22778563104905025,"train/train/tensor_act_model_layers_27_self_attn_v_proj/max_abs":1.09375,"train/train/tensor_grad_model_layers_45_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/mean":-4.9639493227005005e-06,"train/train/tensor_grad_model_layers_26_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_18/act/std":0.41452358450073046,"train/train/tensor_param_model_layers_77_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_mlp/std":0.015580831496054581,"train/train/tensor_act_model_layers_39_mlp_up_proj/mean":-0.0107879638671875,"train/train/layer_model_layers_66/act/norm":9220.161414342878,"train/train/tensor_grad_model_layers_3_mlp_down_proj_weight/norm":0.14031439493169912,"train/train/tensor_act_model_layers_54_self_attn_o_proj/norm":286.6537402911134,"train/train/tensor_act_model_layers_40_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/max_abs":0.0042724609375,"train/train/tensor_act_model_layers_44_mlp_gate_proj/std":0.22754221683531353,"train/train/tensor_act_model_layers_73_mlp_gate_proj/mean":0.00690460205078125,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_gate_proj_weight/mean":-3.2186508178710938e-06,"train/train/tensor_act_model_layers_75_mlp/norm":91.84183844950404,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_self_attn_v_proj/mean":0.019134521484375,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_q_proj_weight/max_abs":0.09326171875,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_42_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_31_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/std":0.02001953125,"train/train/layer__model_layers_40/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15/mean":-0.011749267578125,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/std":0.0004171899520466374,"train/train/tensor_act_model_layers_31_self_attn/mean":0.0021915435791015625,"train/train/tensor_act_model_layers_6_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp_down_proj/norm":89.04695433206945,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34/norm":1671.0293271904995,"train/train/tensor_act_model_layers_92_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_o_proj_weight/std":0.0010171239024853568,"train/train/tensor_act_model_layers_43_mlp/std":0.015580080258191584,"train/train/tensor_grad_model_layers_64_mlp_gate_proj_weight/norm":0.029853991403928203,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_6_self_attn/std":0.04437365704374491,"train/train/tensor_act_model_layers_31_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/std":4.5621232702837106e-07,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/mean":-3.241235390305519e-06,"train/train/tensor_act_model_layers_85/norm":2696.6576989561204,"train/train/tensor_param_model_layers_76_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_self_attn_q_proj_weight/max_abs":4.26173210144043e-06,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42/mean":-0.0034637451171875,"train/train/tensor_grad_model_layers_54_self_attn_o_proj_weight/max_abs":0.003814697265625,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/norm":0.10010271757402166,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/norm":0.10482649045685669,"train/train/tensor_act_model_layers_27_mlp_up_proj/max_abs":1.09375,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/max_abs":8.881092071533203e-06,"train/train/tensor_act_model_layers_23_self_attn_q_proj/norm":1305.7933195974056,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/mean":-2.5937333703041077e-07,"train/train/tensor_act_model_layers_53_input_layernorm/std":1.0000083615009898,"train/train/tensor_param_model_layers_46_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/mean":-5.699694156646729e-07,"train/train/tensor_act_model_layers_57_mlp/norm":90.89496070735242,"train/train/tensor_act_model_layers_41_mlp/std":0.015564372721170703,"train/train/tensor_param_model_layers_24_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_40_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_6_self_attn_k_proj/max_abs":1.09375,"train/train/tensor_act_model_layers_76/mean":0.001758575439453125,"train/train/layer_model_layers_55/act/std":0.42198184413626855,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_79_input_layernorm/mean":0.0019588470458984375,"train/train/layer__model_layers_38/param/norm":17.939248880335683,"train/train/tensor_param_model_layers_47_mlp_gate_proj_weight/max_abs":0.08154296875,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_up_proj/mean":0.004974365234375,"train/train/tensor_grad_model_layers_2_post_attention_layernorm_weight/norm":0.005463157928222673,"train/train/tensor_act_model_layers_79_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_q_proj/mean":-0.025909423828125,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_input_layernorm/norm":5792.583618166515,"train/train/tensor_act_model_layers_46_input_layernorm/max_abs":4.875,"train/train/tensor_act_model_layers_55_self_attn/max_abs":0.2431640625,"train/train/tensor_act_model_layers_77_mlp_up_proj/std":0.22827403021224071,"train/train/tensor_act_model_layers_77_post_attention_layernorm/mean":0.0026617050170898438,"train/train/layer_model_layers_36/act/max_abs":5.09375,"train/train/tensor_act_model_layers_62_input_layernorm/mean":-0.005032539367675781,"train/train/tensor_param_model_layers_51_mlp_gate_proj_weight/mean":-0.00016117095947265625,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/std":0.00022248664128786971,"train/train/tensor_act_model_layers_1_self_attn_v_proj/mean":0.00388336181640625,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_mlp_gate_proj/max_abs":1.1171875,"train/train/tensor_grad_model_layers_78_self_attn_v_proj_weight/std":0.0003894712157416971,"train/train/layer_model_layers_77/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_v_proj/mean":-0.01015472412109375,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/max_abs":0.0751953125,"train/train/tensor_act_model_layers_90_mlp/std":0.015366008009653922,"train/train/layer__model_layers_59/param/std":0.04425919833957362,"train/train/tensor_param_model_layers_16_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_post_attention_layernorm/norm":5792.5968017589375,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/max_abs":0.08349609375,"train/train/tensor_grad_model_layers_29_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/mean":-2.8114300221204758e-08,"train/train/tensor_act_model_layers_55_mlp_gate_proj/std":0.22070635285802143,"train/train/tensor_act_model_layers_42_self_attn_k_proj/mean":0.00445556640625,"train/train/tensor_param_model_layers_71_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_59_self_attn_v_proj_weight/max_abs":0.0791015625,"train/train/tensor_act_model_layers_84_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_2_mlp_down_proj/std":0.015213279036291365,"train/train/tensor_act_model_layers_8_mlp/mean":-0.0003509521484375,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_22/std":0.23194085109800436,"train/train/tensor_act_model_layers_16_self_attn_q_proj/norm":1261.2738632750447,"train/train/layer_model_layers_72/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_1/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_83/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/mean":2.658367156982422e-05,"train/train/tensor_param_model_layers_88_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_64_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_60_input_layernorm/std":1.0000064162317064,"train/train/tensor_act_model_layers_11_mlp_down_proj/norm":91.45940642016465,"train/train/tensor_param_model_layers_56_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_33_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn_k_proj/max_abs":1.140625,"train/train/tensor_act_model_layers_68_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_25_self_attn_o_proj_weight/max_abs":0.0045166015625,"train/train/tensor_act_model_layers_31_self_attn_v_proj/mean":0.0092010498046875,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/std":4.0832572136235884e-07,"train/train/tensor_act_model_layers_73_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_0_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/mean":1.6549602150917053e-06,"train/train/tensor_param_model_layers_76_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_gate_proj/norm":1840.862786256953,"train/train/tensor_param_model_layers_11_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_gate_proj_weight/std":7.489912530250348e-05,"train/train/tensor_act_model_layers_67_self_attn_q_proj/max_abs":1.0546875,"train/train/tensor_act_model_layers_40_self_attn/mean":0.001842498779296875,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/max_abs":0.000949859619140625,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn_v_proj/max_abs":1.203125,"train/train/tensor_act_model_layers_91_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_58/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp_up_proj/norm":1807.1640848977659,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_q_proj/max_abs":1.2890625,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/norm":0.00019625716100669566,"train/train/layer_model_layers_35/grad/norm":0.22801070898068915,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/std":8.312816369751321e-05,"train/train/tensor_param_model_layers_45_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_41/max_abs":1.640625,"train/train/layer_model_layers_7/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_mlp_gate_proj_weight/max_abs":0.00095367431640625,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/mean":-4.667413122660946e-09,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_28_post_attention_layernorm/norm":5792.566528332366,"train/train/tensor_act_model_layers_7_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_q_proj/max_abs":1.0390625,"train/train/tensor_grad_model_layers_31_mlp_gate_proj_weight/max_abs":0.0018157958984375,"train/train/tensor_act_model_layers_48_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/std":0.0004119133776854869,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/max_abs":7.063150405883789e-06,"train/train/tensor_act_model_layers_37_self_attn/max_abs":0.228515625,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/norm":3.609375,"train/train/layer_model_layers_42/act/max_abs":4.90625,"train/train/tensor_grad_model_layers_92_mlp_gate_proj_weight/norm":0.025040932762533157,"train/train/tensor_act_model_layers_92_mlp_down_proj/max_abs":0.0869140625,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_k_proj/norm":1300.5750951033779,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/mean":6.008148193359375e-05,"train/train/tensor_grad_model_layers_79_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_41_mlp_down_proj/max_abs":0.08251953125,"train/train/tensor_act_model_layers_63_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_mlp_down_proj_weight/std":0.00011299351161668219,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/mean":5.900859832763672e-06,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/max_abs":0.0006103515625,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/norm":0.09272556073432435,"train/train/tensor_act_model_layers_67_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/mean":3.956258296966553e-06,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/mean":4.9709342420101166e-08,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_58/grad/max_abs":0.005706787109375,"train/train/tensor_act_model_layers_26_input_layernorm/norm":5792.554931644231,"train/train/tensor_param_model_layers_37_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_input_layernorm/max_abs":4.6875,"train/train/layer_model_layers_53/act/std":0.42199852541937216,"train/train/tensor_act_model_layers_32_post_attention_layernorm/norm":5792.573608399254,"train/train/tensor_act_model_layers_2_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_input_layernorm_weight/max_abs":0.0003986358642578125,"train/train/tensor_act_model_layers_52_self_attn_q_proj/std":0.2209513978677094,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_8_self_attn_o_proj/mean":0.003162384033203125,"train/train/tensor_act_model_layers_73_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn_v_proj/norm":1333.18957249357,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/max_abs":0.09130859375,"train/train/tensor_act_model_layers_52/frac_near_user_limit":0,"train/train/layer_model_layers_21/grad/mean":1.3254010612537355e-06,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_mlp_gate_proj_weight/mean":0.00021648406982421875,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/mean":7.033348083496094e-06,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/max_abs":0.08154296875,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_78_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/std":8.405439892830126e-05,"train/train/tensor_act_model_layers_29_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_input_layernorm/std":1.0000035315689668,"train/train/tensor_param_model_layers_73_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_65_self_attn_o_proj/max_abs":0.2109375,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_16_self_attn_k_proj/std":0.23071646586914607,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/max_abs":0.0012054443359375,"train/train/tensor_param_model_layers_62_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_19/grad/norm":0.33052677079509846,"train/train/tensor_act_model_layers_11_mlp/std":0.015778411519913788,"train/train/tensor_grad_model_layers_2_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_q_proj/mean":0.010833740234375,"train/train/tensor_act_model_layers_81_mlp_up_proj/max_abs":1.0703125,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_72_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_input_layernorm/mean":-0.0035371780395507812,"train/train/tensor_act_model_layers_9_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_7_self_attn_v_proj/mean":-0.0153961181640625,"train/train/tensor_act_model_layers_41_input_layernorm/mean":-0.008892059326171875,"train/train/tensor_act_model_layers_93_self_attn_o_proj/std":0.05041670143461329,"train/train/tensor_act_model_layers_43_self_attn_q_proj/norm":1287.2968947861634,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/max_abs":0.005462646484375,"train/train/tensor_act_model_layers_13/mean":-0.0090484619140625,"train/train/tensor_act_model_layers_30_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/std":7.775594132795835e-05,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_74/norm":2508.411586464004,"train/train/tensor_param_model_layers_25_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_68_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/mean":-5.386769771575928e-06,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/max_abs":0.09716796875,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/max_abs":0.00141143798828125,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_29_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_post_attention_layernorm/mean":-0.05242919921875,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/max_abs":0.21484375,"train/train/tensor_act_model_layers_26_mlp/std":0.01548832222330631,"train/train/tensor_act_model_layers_72_mlp_gate_proj/max_abs":1.109375,"train/train/layer__model_layers_36/param/std":0.04426996702339626,"train/train/layer_model_layers_54/act/norm":9144.098985749953,"train/train/tensor_act_model_layers_46_self_attn_q_proj/norm":1297.6302238499413,"train/train/tensor_act_model_layers_26_self_attn/mean":0.0011844635009765625,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39/max_abs":1.6171875,"train/train/tensor_act_model_layers_25_self_attn_o_proj/norm":263.95461867047135,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/mean":3.241002559661865e-07,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_13_self_attn_v_proj/std":0.22143919856583916,"train/train/layer_model_layers_6/act/std":0.41028211524501207,"train/train/tensor_act_model_layers_7_mlp_up_proj/std":0.23071498421090922,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_q_proj/norm":1302.649226176643,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_mlp/max_abs":0.0859375,"train/train/tensor_grad_model_layers_59_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp_down_proj/std":0.015931039711163653,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/norm":0.00015096019796174363,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_k_proj/norm":1253.9404301809031,"train/train/tensor_param_model_layers_84_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/mean":6.531990948133171e-09,"train/train/tensor_act_model_layers_54_post_attention_layernorm/mean":-0.01079559326171875,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/max_abs":0.005340576171875,"train/train/tensor_act_model_layers_29_mlp_gate_proj/norm":1817.8426907234075,"train/train/tensor_param_model_layers_46_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/norm":0.000868382129647763,"train/train/tensor_grad_model_layers_61_self_attn_k_proj_weight/max_abs":6.3478946685791016e-06,"train/train/layer__model_layers_56/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_q_proj/max_abs":1.0390625,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/max_abs":0.09033203125,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69/max_abs":1.96875,"train/train/tensor_act_model_layers_64_input_layernorm/norm":5792.595092774624,"train/train/tensor_act_model_layers_14_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_33/param/std":0.04423638227916831,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/norm":2.578125,"train/train/layer_model_layers_76/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_mlp_gate_proj_weight/max_abs":0.0888671875,"train/train/tensor_param_model_layers_35_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_16_mlp_up_proj_weight/max_abs":0.0849609375,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_29_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/norm":0.007535756552451947,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/norm":2.59375,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/max_abs":0.00579833984375,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/max_abs":3.7997961044311523e-06,"train/train/tensor_act_model_layers_10_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp_down_proj/std":0.015381691709915752,"train/train/tensor_act_model_layers_8_self_attn_q_proj/norm":1283.3931878277624,"train/train/tensor_param_model_layers_65_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/max_abs":0.0042724609375,"train/train/layer__model_layers_74/param/max_abs":1,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_20/param/max_abs":1,"train/train/tensor_act_model_layers_38_self_attn_q_proj/norm":1309.2506465777205,"train/train/tensor_act_model_layers_62_mlp_down_proj/mean":0.00022920966148376465,"train/train/tensor_act_model_layers_93/std":0.484870521389552,"train/train/tensor_act_model_layers_4_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_mlp_gate_proj_weight/std":7.299278271807142e-05,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/max_abs":0.08642578125,"train/train/tensor_grad_model_layers_38_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_36/param/norm":17.941766424236913,"train/train/layer_model_layers_17/act/max_abs":4.65625,"train/train/tensor_param_model_layers_38_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_o_proj_weight/std":0.0004814734170673136,"train/train/tensor_grad_model_layers_45_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_83_self_attn_v_proj/max_abs":1.0859375,"train/train/tensor_act_model_layers_4_self_attn/max_abs":0.2275390625,"train/train/tensor_act_model_layers_38_mlp_up_proj/std":0.22485831785717275,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/mean":4.909932613372803e-06,"train/train/tensor_act_model_layers_53/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_v_proj_weight/norm":0.19127790858268912,"train/train/layer_model_layers_18/grad/max_abs":0.00787353515625,"train/train/tensor_param_model_layers_21_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_42_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32_self_attn_v_proj/max_abs":1.21875,"train/train/tensor_act_model_layers_33_post_attention_layernorm/max_abs":5.0625,"train/train/tensor_act_model_layers_89/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/std":3.594657797081703e-07,"train/train/tensor_act_model_layers_11_mlp_gate_proj/norm":1859.7630327845443,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/std":0.02001953125,"train/train/layer_model_layers_69/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_38_post_attention_layernorm/mean":-0.0182647705078125,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/mean":1,"train/train/layer_model_layers_67/act/norm":9225.529014085954,"train/train/tensor_act_model_layers_32_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp_gate_proj/std":0.2270530413443356,"train/train/tensor_grad_model_layers_4_mlp_down_proj_weight/max_abs":0.0037994384765625,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_mlp_down_proj/max_abs":0.08837890625,"train/train/tensor_act_model_layers_76_mlp_up_proj/mean":-0.00568389892578125,"train/train/tensor_param_model_layers_63_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_66_self_attn_k_proj_weight/std":4.873002505747241e-07,"train/train/tensor_act_model_layers_86_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_self_attn_v_proj_weight/mean":0.000286102294921875,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/std":3.625309390547645e-07,"train/train/tensor_act_model_layers_71/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/mean":-0.0968017578125,"train/train/layer_model_layers_59/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/mean":-8.058547973632812e-05,"train/train/layer_model_layers_14/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/std":6.910319377998537e-07,"train/train/tensor_act_model_layers_87_self_attn_k_proj/mean":0.01056671142578125,"train/train/tensor_param_model_layers_67_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/mean":0.00011444091796875,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/max_abs":4.1425228118896484e-06,"train/train/tensor_act_model_layers_48_mlp_up_proj/std":0.2226589969981304,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/mean":-0.0002613067626953125,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/mean":-2.5938788894563913e-09,"train/train/layer_model_layers_9/act/mean":-0.010700089590890067,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/max_abs":0.000118255615234375,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/mean":-5.6461431086063385e-08,"train/train/tensor_grad_model_layers_17_mlp_gate_proj_weight/mean":-2.6455381885170937e-08,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/norm":0.00010450623636607764,"train/train/tensor_act_model_layers_9_mlp_up_proj/norm":1829.1339661977036,"train/train/tensor_act_model_layers_47_self_attn_o_proj/mean":-0.000396728515625,"train/train/tensor_act_model_layers_70_mlp/mean":0.00025397539138793945,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/norm":0.09675179306852714,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_56_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_o_proj/mean":-0.0006258487701416016,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_62_self_attn_k_proj/norm":1319.6129198911422,"train/train/tensor_act_model_layers_11_post_attention_layernorm/norm":5792.491699220005,"train/train/tensor_act_model_layers_16_mlp_gate_proj/std":0.22974160819815517,"train/train/tensor_grad_model_layers_47_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_self_attn_k_proj/std":0.2282740710106833,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/std":0.02001953125,"train/train/layer_model_layers_13/act/norm":8958.661738641651,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/std":7.666382385154453e-07,"train/train/tensor_act_model_layers_0/norm":157.31669561260733,"train/train/tensor_act_model_layers_87_self_attn_o_proj/std":0.049868072169304764,"train/train/tensor_act_model_layers_71_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/max_abs":0.00119781494140625,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/norm":0.17656279707351943,"train/train/tensor_act_model_layers_5_mlp_gate_proj/norm":1892.7338050282033,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/norm":2.59375,"train/train/tensor_grad_model_layers_85_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_o_proj/max_abs":0.26171875,"train/train/tensor_act_model_layers_7_mlp/std":0.015686253256347822,"train/train/layer_model_layers_50/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_17_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn/std":0.04895238454512014,"train/train/layer__model_layers_7/param/std":0.04424900284887982,"train/train/tensor_act_model_layers_89_post_attention_layernorm/max_abs":4.59375,"train/train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_60_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_54_mlp_gate_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_6_mlp_up_proj/max_abs":1.328125,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_o_proj/max_abs":0.2158203125,"train/train/tensor_grad_model_layers_88_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_58/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67/norm":2383.7137651530825,"train/train/tensor_act_model_layers_9_self_attn_o_proj/mean":-0.0008592605590820312,"train/train/tensor_grad_model_layers_85_self_attn_q_proj_weight/max_abs":8.58306884765625e-06,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_post_attention_layernorm/max_abs":4.84375,"train/train/tensor_act_model_layers_74_mlp/mean":0.0005035400390625,"train/train/tensor_act_model_layers_8_self_attn_q_proj/max_abs":1.203125,"train/train/tensor_param_model_layers_43_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/max_abs":0.000400543212890625,"train/train/tensor_act_model_layers_79_input_layernorm/norm":5792.59594726881,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_k_proj_weight/max_abs":6.5267086029052734e-06,"train/train/tensor_act_model_layers_28/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25_post_attention_layernorm/max_abs":4.46875,"train/train/tensor_act_model_layers_62_self_attn_q_proj/norm":1335.6589511271334,"train/train/tensor_act_model_layers_69_self_attn_o_proj/norm":281.2734756313297,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/std":3.430955292217577e-05,"train/train/tensor_grad_model_layers_1_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp_down_proj/max_abs":0.087890625,"_step":1,"train/train/tensor_grad_model_layers_86_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_30_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_k_proj_weight/max_abs":1.0788440704345703e-05,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/mean":-1.550006345496513e-09,"train/train/tensor_grad_model_layers_17_mlp_gate_proj_weight/norm":0.061204598785723305,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_43_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/mean":2.630986273288727e-07,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/max_abs":1.5139579772949219e-05,"train/train/layer__model_layers_14/param/std":0.04424273479540164,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/norm":0.005290268599903609,"train/train/tensor_grad_model_layers_52_self_attn_q_proj_weight/norm":0.00013189095131882424,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_post_attention_layernorm_weight/std":2.8845103089425484e-05,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/mean":1.0551884770393372e-06,"train/train/layer_model_layers_40/act/max_abs":4.875,"train/train/tensor_param_model_layers_42_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/mean":5.269050598144531e-05,"train/train/tensor_act_model_layers_82_input_layernorm/std":1.0000008629573045,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/mean":-3.600120544433594e-05,"train/train/layer_model_layers_28/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_k_proj/std":0.22681137696336567,"train/train/layer_model_layers_7/grad/mean":-2.047372110153137e-06,"train/train/layer__model_layers_76/param/frac_near_user_limit":0,"train/train/layer_model_layers_56/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp_down_proj/mean":-0.00038814544677734375,"train/train/layer__model_layers_8/param/mean":0.0014613444645207683,"train/train/tensor_act_model_layers_55_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/mean":-0.000141143798828125,"train/train/tensor_act_model_layers_46_input_layernorm/mean":-0.0067996978759765625,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp_gate_proj/norm":1846.3740155402436,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/mean":-0.0001678466796875,"train/train/layer_model_layers_92/act/max_abs":4.59375,"train/train/tensor_act_model_layers_62_input_layernorm/norm":5792.589599616068,"train/train/tensor_act_model_layers_62/norm":2272.7162100786372,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/mean":-5.209585651755333e-08,"train/train/layer__model_layers_68/param/mean":0.0015409502335903982,"train/train/tensor_act_model_layers_88_post_attention_layernorm/std":1.0000039735671815,"train/train/layer_model_layers_53/grad/mean":2.4647766131436956e-08,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/mean":-1.2576580047607422e-05,"train/train/layer_model_layers_79/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_51_mlp_gate_proj/norm":1871.490732015621,"train/train/tensor_act_model_layers_88/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/max_abs":0.078125,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/mean":2.765655517578125e-05,"train/train/tensor_act_model_layers_40_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/mean":5.878973752260208e-09,"train/train/tensor_act_model_layers_86_self_attn_o_proj/norm":281.20053303639503,"train/train/tensor_grad_model_layers_17_self_attn_k_proj_weight/norm":0.00043031006822761626,"train/train/tensor_act_model_layers_26_self_attn/std":0.0496834458975016,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/std":0.00010117259760920237,"train/train/tensor_param_model_layers_69_mlp_gate_proj_weight/norm":3.625,"train/train/layer_model_layers_82/act/mean":1.821028334753854e-05,"train/train/tensor_act_model_layers_59_mlp/max_abs":0.087890625,"train/train/tensor_act_model_layers_42_mlp_down_proj/mean":-0.0003559589385986328,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/mean":-0.000179290771484375,"train/train/tensor_act_model_layers_41_mlp_up_proj/frac_near_user_limit":0,"train/train/layer_model_layers_72/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_mlp_down_proj/max_abs":0.08642578125,"train/train/tensor_grad_model_layers_3_input_layernorm_weight/norm":0.012140972689703064,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_mlp/std":0.015931039711163653,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_k_proj_weight/norm":0.0014157615631013756,"train/train/tensor_act_model_layers_61_mlp_down_proj/std":0.016174759819406002,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_33_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn/max_abs":0.234375,"train/train/tensor_act_model_layers_1_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/mean":0.00024127960205078125,"train/train/layer__model_layers_63/param/std":0.04423094640994518,"train/train/layer__model_layers_24/param/max_abs":1,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_input_layernorm_weight/mean":-4.8041343688964844e-05,"train/train/tensor_act_model_layers_23_self_attn_k_proj/norm":1296.1167348159963,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_91_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_9_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/max_abs":0.09228515625,"train/train/tensor_act_model_layers_82_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_59_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_45/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/norm":0.0015128677678049802,"train/train/tensor_act_model_layers_64_mlp_down_proj/norm":90.59842211152406,"train/train/tensor_act_model_layers_1_self_attn_k_proj/mean":0.0013704299926757812,"train/train/tensor_act_model_layers_43_self_attn_v_proj/std":0.21850968709182347,"train/train/tensor_act_model_layers_13_self_attn/mean":-0.0009279251098632812,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/max_abs":1.0788440704345703e-05,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/mean":-0.00024127960205078125,"train/train/tensor_grad_model_layers_12_input_layernorm_weight/max_abs":0.00164031982421875,"train/train/layer__model_layers_8/param/norm":17.933654585635217,"train/train/tensor_act_model_layers_8_mlp_gate_proj/mean":0.00013715028762817383,"train/train/tensor_act_model_layers_17_self_attn_v_proj/norm":1327.8954705799415,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/mean":1.6975718608591706e-09,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/norm":0.00016153382190795175,"train/train/tensor_act_model_layers_70_self_attn/norm":283.6257155733301,"train/train/tensor_grad_model_layers_76_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_44_mlp_gate_proj/mean":0.0007582902908325195,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_3_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_74_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_57_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/norm":0.00023006243998250104,"train/train/tensor_param_model_layers_16_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/max_abs":0.08837890625,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp_up_proj/std":0.22876064547539152,"train/train/tensor_grad_model_layers_14_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_mlp_up_proj/mean":3.471970558166504e-05,"train/train/tensor_act_model_layers_79_mlp/std":0.015488971304149621,"train/train/tensor_grad_model_layers_23_self_attn_k_proj_weight/mean":4.640440920411493e-09,"train/train/tensor_act_model_layers_17/norm":1160.696803131127,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/mean":1.1026859283447266e-05,"train/train/tensor_grad_model_layers_5_post_attention_layernorm_weight/mean":-1.4834105968475342e-05,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/mean":8.307397365570068e-07,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_30_self_attn_o_proj/mean":0.0004100799560546875,"train/train/tensor_act_model_layers_61/norm":2242.23788204883,"train/train/tensor_act_model_layers_74_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35/std":0.2919953430405561,"train/train/tensor_act_model_layers_43/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn/max_abs":0.234375,"train/train/layer_model_layers_55/act/max_abs":5.0625,"train/train/tensor_act_model_layers_66_self_attn_o_proj/mean":-0.0017528533935546875,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/std":1.0580758520905213e-06,"train/train/tensor_act_model_layers_65/std":0.4018659832047066,"train/train/tensor_param_model_layers_66_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/norm":0.11790930658741718,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/norm":0.0021960334501873345,"train/train/tensor_act_model_layers_69_mlp/std":0.015442409488830584,"train/train/tensor_act_model_layers_78_mlp_down_proj/mean":2.0682811737060547e-05,"train/train/tensor_grad_model_layers_4_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_mlp_up_proj/std":0.22363592521866388,"train/train/layer_model_layers_46/grad/norm":0.1855978927464571,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/mean":-5.948822945356369e-08,"train/train/tensor_act_model_layers_13_mlp/std":0.015810022945638966,"train/train/tensor_act_model_layers_72_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/mean":2.3096799850463867e-06,"train/train/layer__model_layers_27/param/norm":17.924681013985296,"train/train/tensor_act_model_layers_25_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_q_proj/std":0.23218449360628607,"train/train/tensor_act_model_layers_65_mlp/max_abs":0.08251953125,"train/train/tensor_grad_model_layers_44_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_k_proj/norm":1336.2329789974924,"train/train/tensor_act_model_layers_62_mlp/std":0.015503561349066197,"train/train/layer_model_layers_81/grad/norm":0.1478312987596523,"train/train/layer_model_layers_67/act/max_abs":4.5,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/std":3.6090171675107003e-07,"train/train/tensor_act_model_layers_58_self_attn_v_proj/std":0.2312042044538385,"train/train/tensor_grad_model_layers_58_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/max_abs":0.08642578125,"train/train/tensor_param_model_layers_89_self_attn_q_proj_weight/mean":-0.00018787384033203125,"train/train/tensor_param_model_layers_17_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/std":4.831704830946435e-05,"train/train/tensor_param_model_layers_24_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/max_abs":0.0147705078125,"train/train/tensor_act_model_layers_28_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer_model_layers_86/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/max_abs":0.07958984375,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/norm":0.0007734786699095452,"train/train/tensor_act_model_layers_56_self_attn_v_proj/norm":1301.709784521809,"train/train/layer_model_layers_55/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/max_abs":0.000499725341796875,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/max_abs":0.08203125,"train/train/tensor_act_model_layers_9_self_attn_q_proj/norm":1299.7677877678573,"train/train/tensor_act_model_layers_73_self_attn/max_abs":0.244140625,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/std":8.55577483702662e-05,"train/train/tensor_act_model_layers_78_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_mlp_down_proj_weight/max_abs":0.00177764892578125,"train/train/tensor_act_model_layers_36_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_37/act/mean":-0.0021604938166482107,"train/train/tensor_act_model_layers_49_self_attn_q_proj/max_abs":1.1171875,"train/train/tensor_grad_model_layers_92_mlp_gate_proj_weight/mean":-1.3329554349184036e-07,"train/train/tensor_grad_model_layers_60_mlp_up_proj_weight/max_abs":0.001129150390625,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_q_proj/max_abs":0.9921875,"train/train/tensor_act_model_layers_50_self_attn_k_proj/mean":0.01947021484375,"train/train/tensor_act_model_layers_36_mlp_up_proj/std":0.2324263042340463,"train/train/tensor_act_model_layers_81_self_attn_k_proj/std":0.22339247630244646,"train/train/layer_model_layers_42/grad/max_abs":0.005706787109375,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/std":0.0029539618544133016,"train/train/tensor_act_model_layers_74_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_68_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/mean":-0.00017070770263671875,"train/train/tensor_grad_model_layers_41_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_5_mlp_up_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/std":0.00019493699472461918,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_mlp_gate_proj/norm":1885.3274121492013,"train/train/tensor_act_model_layers_7_mlp_up_proj/norm":1889.5370853077757,"train/train/tensor_grad_model_layers_67_self_attn_k_proj_weight/max_abs":4.082918167114258e-06,"train/train/tensor_act_model_layers_86/frac_near_dtype_limit":0,"train/train/layer__model_layers_67/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn/max_abs":0.2294921875,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_52_self_attn_o_proj/std":0.0494399258447518,"train/train/tensor_act_model_layers_13_mlp_gate_proj/mean":0.011016845703125,"train/train/tensor_act_model_layers_44_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_7_mlp_up_proj_weight/mean":-9.117648005485535e-07,"train/train/tensor_act_model_layers_7_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_93_mlp/max_abs":0.0830078125,"train/train/tensor_act_model_layers_85/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_self_attn_o_proj/norm":280.3629927107138,"train/train/tensor_grad_model_layers_7_self_attn_v_proj_weight/norm":0.3642918466003758,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/max_abs":0.0004730224609375,"train/train/tensor_act_model_layers_68_mlp/norm":92.68890968088337,"train/train/tensor_grad_model_layers_88_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/norm":0.11952073792751938,"train/train/tensor_act_model_layers_1_mlp/norm":90.17691290225004,"train/train/tensor_grad_model_layers_49_self_attn_v_proj_weight/mean":-4.4330954551696777e-07,"train/train/tensor_act_model_layers_61_self_attn/norm":281.2532324693673,"train/train/layer_model_layers_30/grad/norm":0.23533142440202814,"train/train/tensor_grad_model_layers_23_mlp_gate_proj_weight/max_abs":0.0020599365234375,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/max_abs":0.0036773681640625,"train/train/tensor_act_model_layers_81/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/norm":1314.8780012638813,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_self_attn_q_proj/norm":1321.0643927647118,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/norm":0.03516387856686857,"train/train/tensor_act_model_layers_66_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/max_abs":0.0025787353515625,"train/train/tensor_grad_model_layers_69_mlp_gate_proj_weight/std":7.475596312888152e-05,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/mean":-0.00016117095947265625,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_v_proj/mean":0.0198974609375,"train/train/tensor_act_model_layers_4_input_layernorm/norm":5791.902709964996,"train/train/tensor_act_model_layers_4_mlp_up_proj/mean":0.003173828125,"train/train/tensor_param_model_layers_17_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_16_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_mlp_up_proj/std":0.2270540823686781,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_input_layernorm/max_abs":5.03125,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/max_abs":3.844499588012695e-06,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/std":3.5201388090086187e-07,"train/train/tensor_grad_model_layers_50_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/norm":0.8709601740354488,"train/train/layer_model_layers_15/grad/max_abs":0.0103759765625,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_56_mlp/norm":88.81283999801195,"train/train/tensor_act_model_layers_54_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/norm":0.030590367294817532,"train/train/tensor_act_model_layers_6_post_attention_layernorm/norm":5792.348876958985,"train/train/tensor_act_model_layers_29_post_attention_layernorm/norm":5792.572753906785,"train/train/tensor_act_model_layers_17/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_81/act/std":0.4282826734194672,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/max_abs":0.00052642822265625,"train/train/tensor_grad_model_layers_6_post_attention_layernorm_weight/max_abs":0.0004329681396484375,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/mean":-0.000141143798828125,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0/std":0.027160735930465828,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_self_attn_k_proj/mean":-0.0143280029296875,"train/train/tensor_act_model_layers_35_self_attn_k_proj/max_abs":1.2578125,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_20/act/max_abs":4.46875,"train/train/tensor_act_model_layers_21_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_26/act/max_abs":4.65625,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/std":8.673430534570975e-05,"train/train/tensor_act_model_layers_0_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_46/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/mean":-4.0531158447265625e-05,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/norm":0.17285363147596497,"train/train/tensor_act_model_layers_39_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/mean":-1.312466338276863e-06,"train/train/tensor_act_model_layers_64_input_layernorm/mean":0.000885009765625,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_51_mlp_gate_proj/max_abs":1.1328125,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/norm":0.0023529972993539946,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn/norm":276.3456489040801,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76/std":0.43848659738798745,"train/train/tensor_act_model_layers_56_self_attn_v_proj/std":0.2248593090927135,"train/train/tensor_act_model_layers_59_post_attention_layernorm/norm":5792.594604493992,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_88/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_51_mlp_down_proj/norm":90.24203654185172,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/std":8.478724972150094e-05,"train/train/layer__model_layers_12/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/max_abs":0.0869140625,"train/train/tensor_act_model_layers_35_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/max_abs":0.0003566741943359375,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_self_attn/norm":108.08046583816427,"train/train/tensor_act_model_layers_88_self_attn/max_abs":0.2236328125,"train/train/tensor_act_model_layers_33/std":0.2846729713044551,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/mean":1.816079020500183e-08,"train/train/tensor_act_model_layers_88_mlp_gate_proj/std":0.22607863340923362,"train/train/tensor_act_model_layers_86_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14/std":0.17773800298635015,"train/train/tensor_act_model_layers_70/mean":-0.0009628087282180786,"train/train/tensor_grad_model_layers_64_mlp_gate_proj_weight/mean":-3.8510188460350037e-07,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/norm":0.00035191357611822016,"train/train/tensor_act_model_layers_84_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_v_proj/norm":1325.6699619166175,"train/train/tensor_act_model_layers_43_self_attn_k_proj/norm":1365.2848330909776,"train/train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/mean":-2.8194335754960775e-10,"train/train/tensor_act_model_layers_40/norm":1804.15011459063,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/max_abs":0.08251953125,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/max_abs":0.000293731689453125,"train/train/tensor_param_model_layers_65_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn/std":0.048952948618697076,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/norm":0.00012837019563648037,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_mlp_gate_proj_weight/max_abs":0.0027618408203125,"train/train/tensor_param_model_layers_4_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_o_proj/mean":0.0019893646240234375,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/mean":3.039836883544922e-05,"train/train/tensor_act_model_layers_87_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/norm":0.1471112886973907,"train/train/tensor_act_model_layers_17_mlp_up_proj/max_abs":1.1953125,"train/train/tensor_act_model_layers_88_self_attn_q_proj/mean":-0.0035648345947265625,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/mean":6.51925802230835e-08,"train/train/tensor_act_model_layers_70_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/norm":2.546875,"train/train/layer__model_layers_14/param/norm":17.933212139944423,"train/train/tensor_grad_model_layers_44_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_mlp_up_proj_weight/mean":2.3766187950968742e-07,"train/train/tensor_grad_model_layers_77_input_layernorm_weight/max_abs":0.00057220458984375,"train/train/tensor_param_model_layers_15_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_4_input_layernorm_weight/max_abs":0.00148773193359375,"train/train/tensor_param_model_layers_84_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/mean":0.0002155303955078125,"train/train/layer_model_layers_74/grad/norm":0.16159124401795572,"train/train/tensor_act_model_layers_45_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/std":1.0711079801971996e-06,"train/train/tensor_act_model_layers_79_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp_down_proj/norm":89.49962827904578,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_o_proj_weight/max_abs":0.005706787109375,"train/train/tensor_grad_model_layers_20_self_attn_k_proj_weight/max_abs":8.404254913330078e-06,"train/train/tensor_act_model_layers_19_mlp/max_abs":0.08203125,"train/train/tensor_act_model_layers_19_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/max_abs":0.080078125,"train/train/tensor_param_model_layers_78_self_attn_o_proj_weight/max_abs":0.0830078125,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_24_self_attn_q_proj/std":0.22217390198566261,"train/train/tensor_grad_model_layers_37_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_51_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44/max_abs":1.6328125,"train/train/tensor_act_model_layers_75_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_gate_proj_weight/mean":0.0001735687255859375,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_k_proj/mean":0.0011687278747558594,"train/train/tensor_act_model_layers_1_self_attn_k_proj/max_abs":1.1796875,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_mlp/norm":89.04695433206945,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/mean":9.119510650634766e-06,"train/train/tensor_act_model_layers_59_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_36/act/norm":9099.451080085932,"train/train/tensor_param_model_layers_16_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/max_abs":0.00101470947265625,"train/train/tensor_param_model_layers_30_self_attn_q_proj_weight/max_abs":0.08642578125,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_k_proj/max_abs":1.09375,"train/train/tensor_act_model_layers_54_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/std":0.00011097507787500258,"train/train/tensor_grad_model_layers_84_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_mlp_gate_proj_weight/norm":0.04159482086431038,"train/train/tensor_act_model_layers_15_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_26_input_layernorm/mean":-0.04962158203125,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_gate_proj/norm":1852.1982509828106,"train/train/tensor_act_model_layers_38_self_attn_k_proj/std":0.22241397484207448,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/mean":9.393692016601562e-05,"train/train/tensor_act_model_layers_70_self_attn_o_proj/mean":-0.0005165338516235352,"train/train/tensor_param_model_layers_71_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_31_self_attn/norm":274.9124768963415,"train/train/tensor_grad_model_layers_62_mlp_down_proj_weight/std":7.832809214954274e-05,"train/train/tensor_act_model_layers_13_self_attn_v_proj/mean":0.00701904296875,"train/train/tensor_act_model_layers_27_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/mean":-9.059906005859375e-05,"train/train/tensor_act_model_layers_46_self_attn_k_proj/norm":1284.9016162290125,"train/train/layer__model_layers_26/param/mean":0.0016015301256581513,"train/train/layer_model_layers_61/grad/max_abs":0.004852294921875,"train/train/layer_model_layers_57/grad/mean":-2.798177563409537e-07,"train/train/tensor_act_model_layers_39_self_attn_o_proj/max_abs":0.23828125,"train/train/tensor_param_model_layers_34_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_k_proj/norm":1293.5683397793728,"train/train/tensor_param_model_layers_12_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_60_input_layernorm/max_abs":4.5625,"train/train/layer__model_layers_85/param/norm":17.936390699584045,"train/train/tensor_act_model_layers_93_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_self_attn/mean":-0.0001914501190185547,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp/max_abs":0.08447265625,"train/train/tensor_act_model_layers_5_input_layernorm/std":1.0000108386733848,"train/train/tensor_act_model_layers_87_self_attn_q_proj/std":0.22315281366382292,"train/train/tensor_act_model_layers_66_mlp_down_proj/max_abs":0.076171875,"train/train/layer__model_layers_17/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_q_proj/norm":1282.6697587234837,"train/train/tensor_act_model_layers_69_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/norm":0.0007944421887885316,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp_down_proj/norm":90.89496070735242,"train/train/layer_model_layers_33/grad/mean":-3.628307799467245e-08,"train/train/tensor_act_model_rotary_emb/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/mean":4.076957702636719e-05,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/std":0.0005860973010961975,"train/train/tensor_grad_model_layers_21_mlp_gate_proj_weight/max_abs":0.00193023681640625,"train/train/tensor_act_model_layers_54_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/std":0.0006956944059005876,"train/train/tensor_param_model_layers_75_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_mlp_up_proj_weight/mean":0.0001354217529296875,"train/train/tensor_param_model_layers_53_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_10_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_34_self_attn_q_proj/max_abs":1.0546875,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer_model_layers_33/grad/std":0.00029070832837382036,"train/train/tensor_act_model_layers_1_input_layernorm/norm":5788.557861342629,"train/train/tensor_act_model_layers_45_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn_o_proj/max_abs":0.2158203125,"train/train/tensor_grad_model_layers_56_mlp_gate_proj_weight/max_abs":0.00124359130859375,"train/train/tensor_grad_model_layers_25_mlp_gate_proj_weight/norm":0.0424904647116006,"train/train/tensor_act_model_layers_75_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/max_abs":1.1622905731201172e-05,"train/train/tensor_act_model_layers_16_self_attn/std":0.05102959267746651,"train/train/tensor_act_model_layers_84_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_input_layernorm/std":1.0000099394907807,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/max_abs":0.08837890625,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_k_proj/norm":1292.393156924987,"train/train/tensor_act_model_layers_55_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_25_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_input_layernorm/norm":5792.578247071181,"train/train/tensor_param_model_layers_70_self_attn_v_proj_weight/mean":-1.8358230590820312e-05,"train/train/tensor_param_model_layers_32_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43/norm":1848.5001866371333,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/max_abs":0.0011749267578125,"train/train/tensor_act_model_layers_66_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_48_self_attn/max_abs":0.220703125,"train/train/tensor_grad_model_layers_78_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/norm":2.5625,"train/train/layer_model_layers_78/act/norm":9272.147868352324,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/max_abs":0.07568359375,"train/train/tensor_act_model_layers_79_mlp/max_abs":0.08154296875,"train/train/tensor_act_model_layers_67_input_layernorm/norm":5792.591430664468,"train/train/tensor_act_model_layers_14_mlp_down_proj/mean":0.00014853477478027344,"train/train/tensor_act_model_layers_2_self_attn_o_proj/mean":0.0016460418701171875,"train/train/tensor_act_model_layers_42_mlp/max_abs":0.0859375,"train/train/tensor_param_model_layers_93_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_mlp/max_abs":0.08154296875,"train/train/tensor_param_model_layers_90_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/mean":-2.605374902486801e-07,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/norm":0.0020152038702141335,"train/train/layer_model_layers_48/grad/max_abs":0.005401611328125,"train/train/tensor_act_model_layers_10/max_abs":0.78125,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/max_abs":0.0033111572265625,"train/train/tensor_act_model_layers_90_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_mlp/max_abs":0.08154296875,"train/train/tensor_param_model_layers_31_mlp_gate_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_7_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_5_input_layernorm/mean":-0.05413818359375,"train/train/tensor_act_model_layers_63_self_attn_q_proj/norm":1287.7704408770883,"train/train/tensor_act_model_layers_17_self_attn_v_proj/max_abs":1.0546875,"train/train/tensor_param_model_layers_13_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/max_abs":0.0072021484375,"train/train/tensor_param_model_layers_19_self_attn_v_proj_weight/max_abs":0.0869140625,"train/train/tensor_param_model_layers_22_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_2_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_4/max_abs":0.53515625,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_mlp_down_proj_weight/max_abs":0.0830078125,"train/train/layer_model_layers_26/act/norm":9017.340713435862,"train/train/tensor_param_model_layers_88_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_53_mlp_gate_proj_weight/mean":1.6808509826660156e-05,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_17_post_attention_layernorm/norm":5792.537719727133,"train/train/tensor_act_model_layers_89_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_o_proj/mean":0.0014495849609375,"train/train/layer_model_layers_90/grad/frac_near_dtype_limit":0,"train/train/layer__model_layers_88/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_input_layernorm/std":1.000003122725174,"train/train/tensor_act_model_layers_40_self_attn/norm":277.3989473045707,"train/train/tensor_act_model_layers_16_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/mean":3.8623809814453125e-05,"train/train/tensor_act_model_layers_72_self_attn_o_proj/std":0.050173352072587184,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_18_self_attn/max_abs":0.2119140625,"train/train/tensor_grad_model_layers_77_mlp_gate_proj_weight/norm":0.02729145225896892,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/max_abs":0.006317138671875,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_76_input_layernorm/max_abs":4.71875,"train/train/tensor_act_model_layers_58_input_layernorm/norm":5792.593872072394,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_10_post_attention_layernorm/max_abs":4.9375,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_38_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_input_layernorm_weight/max_abs":1,"train/train/layer__model_layers_73/param/std":0.04422547306775391,"train/train/layer__model_layers_27/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/std":9.099836234028599e-05,"train/train/layer_model_layers_82/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_mlp_gate_proj/std":0.22436788103221597,"train/train/tensor_param_model_layers_79_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_act_model_norm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_o_proj/std":0.039127129065247804,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/std":6.923634814076211e-05,"train/train/tensor_param_model_layers_28_mlp_gate_proj_weight/mean":-4.76837158203125e-05,"train/train/tensor_act_model_layers_34_self_attn_o_proj/std":0.04852376368689302,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/std":4.4900829573589266e-05,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_11_mlp_down_proj_weight/std":0.0002002814411850371,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_self_attn_o_proj/mean":0.0021915435791015625,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_q_proj/std":0.23022556658694485,"train/train/tensor_param_model_layers_38_mlp_gate_proj_weight/mean":-3.2901763916015625e-05,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/std":3.1858456558933256e-06,"train/train/tensor_param_model_layers_4_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_gate_proj/std":0.2258331355869804,"train/train/tensor_act_model_layers_10_self_attn_o_proj/mean":-0.001552581787109375,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/norm":0.00021769848883492526,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/mean":4.7206878662109375e-05,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/mean":0.00017261505126953125,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/norm":0.08054457860091076,"train/train/tensor_act_model_layers_89_post_attention_layernorm/std":1.0000040990914958,"train/train/tensor_grad_model_layers_25_self_attn_q_proj_weight/mean":3.230525180697441e-09,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_49_self_attn_k_proj_weight/max_abs":0.08154296875,"train/train/layer_model_layers_87/grad/max_abs":0.004150390625,"train/train/tensor_act_model_layers_67_input_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_35/act/norm":9052.34988915104,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_1/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_input_layernorm_weight/norm":0.002463821979974325,"train/train/tensor_param_model_layers_12_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_mlp/norm":90.95295758225211,"train/train/tensor_act_model_layers_57_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_48_mlp_down_proj/mean":0.00019985437393188477,"train/train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/std":6.850299567002901e-05,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/max_abs":8.821487426757812e-06,"train/train/tensor_param_model_layers_48_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_27_mlp_gate_proj_weight/norm":0.04538356864037921,"train/train/tensor_act_model_layers_39_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_post_attention_layernorm/std":1.0000038580836343,"train/train/layer__model_layers_90/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_self_attn_k_proj/max_abs":1.09375,"train/train/tensor_act_model_layers_38_input_layernorm/std":1.000006251446862,"train/train/tensor_act_model_layers_3_mlp_gate_proj/norm":1843.1283907902919,"train/train/tensor_param_model_layers_77_self_attn_q_proj_weight/mean":-8.463859558105469e-06,"train/train/tensor_act_model_layers_60_post_attention_layernorm/norm":5792.594360355312,"train/train/tensor_param_model_layers_72_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77/frac_near_user_limit":0,"train/train/layer_model_layers_67/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75/norm":2514.5967121926906,"train/train/tensor_act_model_layers_50_post_attention_layernorm/norm":5792.586669925132,"train/train/tensor_param_model_layers_0_mlp_up_proj_weight/norm":3.625,"train/train/layer__model_layers_65/param/std":0.04427946387465865,"train/train/tensor_act_model_layers_10_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/max_abs":5.543231964111328e-06,"train/train/layer_model_layers_68/act/max_abs":4.46875,"train/train/tensor_act_model_layers_1_mlp/max_abs":0.09423828125,"train/train/tensor_act_model_layers_8_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/norm":0.04043384475342642,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/max_abs":0.00112152099609375,"train/train/tensor_grad_model_layers_30_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_22_self_attn_q_proj/mean":0.00331878662109375,"train/train/layer_model_layers_65/act/max_abs":4.59375,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/mean":8.344650268554688e-05,"train/train/tensor_act_model_layers_70_post_attention_layernorm/mean":-0.00327301025390625,"train/train/layer_model_layers_70/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_o_proj/std":0.04993066708292556,"train/train/tensor_act_model_layers_92_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_mlp/std":0.015382231767609566,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/max_abs":0.0013427734375,"train/train/tensor_act_model_layers_8/mean":-0.00727081298828125,"train/train/layer__model_layers_63/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_85_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_55_mlp_gate_proj_weight/max_abs":0.076171875,"train/train/tensor_act_model_layers_31/max_abs":1.40625,"train/train/layer_model_layers_54/grad/max_abs":0.00518798828125,"train/train/tensor_param_model_layers_28_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/std":0.00012667012715488095,"train/train/tensor_act_model_layers_17_input_layernorm/mean":-0.06280517578125,"train/train/tensor_param_model_layers_9_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_post_attention_layernorm/std":1.0000015683329524,"train/train/layer_model_layers_26/act/mean":-0.005129473549979073,"train/train/tensor_act_model_layers_49_mlp_gate_proj/max_abs":1.09375,"train/train/tensor_act_model_layers_40_self_attn_o_proj/std":0.04785333281643411,"train/train/tensor_act_model_layers_42_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_input_layernorm/norm":5792.597045913022,"train/train/tensor_act_model_layers_6_mlp_down_proj/max_abs":0.0849609375,"train/train/tensor_param_model_layers_84_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_2_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_76_mlp/max_abs":0.08056640625,"train/train/tensor_act_model_layers_72_self_attn_v_proj/max_abs":1.125,"train/train/tensor_act_model_layers_5_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_self_attn_k_proj/std":0.22608127696263441,"train/train/tensor_act_model_layers_79_self_attn_k_proj/norm":1325.4813901914172,"train/train/tensor_act_model_layers_93_self_attn_v_proj/mean":0.007110595703125,"train/train/tensor_act_model_layers_15_mlp_up_proj/norm":1878.3086132175583,"train/train/tensor_grad_model_layers_82_self_attn_q_proj_weight/std":4.107074985027203e-07,"train/train/tensor_act_model_layers_44_post_attention_layernorm/max_abs":4.90625,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_86_mlp_gate_proj/max_abs":1.0703125,"train/train/tensor_grad_model_layers_70_input_layernorm_weight/mean":4.898756742477417e-06,"train/train/tensor_act_model_layers_47_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_mlp_gate_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_47_self_attn_v_proj/std":0.23267039618115995,"train/train/layer_model_layers_70/grad/max_abs":0.0062255859375,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_86_post_attention_layernorm/max_abs":4.71875,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_40_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_mlp/mean":-0.0003352165222167969,"train/train/tensor_act_model_layers_48_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29_self_attn/mean":-0.00130462646484375,"train/train/layer_model_layers_50/act/std":0.4215737193212493,"train/train/tensor_param_model_layers_79_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_mlp_gate_proj/std":0.229983151354654,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/mean":4.8160552978515625e-05,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_35_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/mean":-2.765655517578125e-05,"train/train/tensor_act_model_layers_33_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_post_attention_layernorm/std":1.0000011380755385,"train/train/tensor_grad_model_layers_89_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_q_proj/norm":1288.7320554275946,"train/train/tensor_param_model_layers_92_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_87_post_attention_layernorm/mean":0.0111083984375,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/mean":5.804002285003662e-06,"train/train/tensor_act_model_layers_33_self_attn_q_proj/std":0.2238789748638223,"train/train/tensor_act_model_layers_35_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/mean":3.4809112548828125e-05,"train/train/tensor_act_model_layers_51_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/std":0.0004152400363739598,"train/train/tensor_act_model_layers_65_mlp/norm":94.09292112202088,"train/train/layer_model_layers_92/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_mlp_down_proj_weight/mean":-0.000156402587890625,"train/train/tensor_grad_model_layers_61_self_attn_o_proj_weight/std":0.0004394751367778309,"train/train/tensor_act_model_layers_24_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_v_proj_weight/max_abs":0.0177001953125,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_37_post_attention_layernorm/max_abs":5.03125,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_51_post_attention_layernorm/std":1.0000082567571744,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/max_abs":4.351139068603516e-06,"train/train/tensor_act_model_layers_7_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_51/act/std":0.4221379805464317,"train/train/layer_model_layers_76/grad/std":0.00019323462140906486,"train/train/tensor_act_model_layers_53_self_attn_v_proj/mean":0.0045623779296875,"train/train/tensor_act_model_layers_52_input_layernorm/mean":-0.0033626556396484375,"train/train/tensor_act_model_layers_40_input_layernorm/norm":5792.5831298847515,"train/train/tensor_act_model_layers_30_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_83/param/norm":17.937302644820235,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_10_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp/mean":0.0007171630859375,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/mean":2.1338462829589844e-05,"train/train/tensor_param_model_layers_55_mlp_gate_proj_weight/mean":0.0001010894775390625,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/mean":1.2097880244255066e-06,"train/train/tensor_act_model_layers_54_self_attn/norm":286.6537402911134,"train/train/tensor_act_model_layers_81_self_attn_o_proj/norm":293.8579930583087,"train/train/tensor_param_model_layers_40_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_58/mean":-0.0018873214721679688,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/max_abs":0.004913330078125,"train/train/tensor_grad_model_layers_33_self_attn_v_proj_weight/norm":0.16372646249982034,"train/train/tensor_act_model_layers_74_self_attn_q_proj/mean":-0.0019788742065429688,"train/train/tensor_param_model_layers_90_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_64_mlp/mean":9.109079837799072e-05,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_input_layernorm/std":1.0000079129923296,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/norm":0.05799367713171312,"train/train/layer_model_layers_51/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_self_attn_v_proj_weight/std":0.0006095002688061062,"train/train/tensor_param_model_layers_15_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_61/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/mean":7.290509529411793e-09,"train/train/tensor_act_model_layers_31_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_84/grad/norm":0.13472647888660458,"train/train/tensor_act_model_layers_51_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_down_proj/norm":90.9719906597224,"train/train/tensor_act_model_layers_69_input_layernorm/norm":5792.604125977513,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/max_abs":0.00011157989501953125,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/std":2.8518542058536356e-05,"train/train/layer_model_layers_90/grad/max_abs":0.004119873046875,"train/train/tensor_grad_model_layers_19_self_attn_q_proj_weight/norm":0.00027420085340159765,"train/train/tensor_act_model_layers_77_self_attn/std":0.05011241332546886,"train/train/tensor_param_model_layers_13_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_50_mlp/norm":92.51323163452432,"train/train/tensor_param_model_layers_52_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49/std":0.34326633249897687,"train/train/tensor_param_model_layers_27_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_down_proj/max_abs":0.08447265625,"train/train/tensor_grad_model_layers_56_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/max_abs":0.07861328125,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/mean":-9.72747802734375e-05,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/mean":6.705522537231445e-08,"train/train/tensor_param_model_layers_1_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_93_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_9_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/max_abs":5.9604644775390625e-06,"train/train/tensor_grad_model_layers_60_post_attention_layernorm_weight/std":3.402126336539529e-05,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_self_attn/max_abs":0.2109375,"train/train/tensor_act_model_layers_66/frac_near_dtype_limit":0,"train/train/layer_model_layers_45/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_84_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/max_abs":5.692243576049805e-06,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/std":4.395584615619753e-05,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/norm":0.11180174619030954,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_input_layernorm/max_abs":4.5625,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/mean":-4.032626748085022e-07,"train/train/tensor_act_model_layers_70_self_attn/max_abs":0.2421875,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/mean":3.0035153031349182e-06,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/norm":0.0001107751320919808,"train/train/layer__model_layers_75/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_k_proj_weight/mean":-4.318280844017863e-09,"train/train/tensor_act_model_layers_92_self_attn_o_proj/std":0.05078515950585719,"train/train/layer__model_layers_75/param/std":0.044284998644662066,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_78_self_attn_v_proj_weight/std":0.0201416015625,"train/train/layer__model_layers_58/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_63/act/norm":9199.290691817678,"train/train/tensor_act_model_layers_74_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/norm":0.00010647224928041016,"train/train/tensor_act_model_layers_39_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/mean":-4.4419721234589815e-09,"train/train/tensor_param_model_layers_17_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_mlp_gate_proj_weight/mean":2.6496127247810364e-07,"train/train/tensor_act_model_layers_17_self_attn_o_proj/norm":271.9793402297586,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/norm":0.0001373017800491815,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/mean":-4.863739013671875e-05,"train/train/global/grad/norm":3.5937337127353737,"train/train/layer__model_layers_52/param/mean":0.0015327346492297193,"train/train/tensor_act_model_layers_34_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_o_proj/max_abs":0.2451171875,"train/train/tensor_act_model_layers_2_self_attn_o_proj/max_abs":0.251953125,"train/train/tensor_param_model_layers_6_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/norm":0.0019293394754645395,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/norm":2.53125,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/mean":-3.896730049746111e-09,"train/train/tensor_act_model_layers_38_post_attention_layernorm/std":1.0000050043329176,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_53_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_mlp_gate_proj_weight/norm":0.027051591000332335,"train/train/tensor_act_model_layers_48_input_layernorm/max_abs":4.6875,"train/train/layer_model_layers_22/grad/norm":0.2764964355136974,"train/train/tensor_act_model_layers_32_mlp/mean":0.00024127960205078125,"train/train/tensor_act_model_layers_54_self_attn/mean":-0.00246429443359375,"train/train/tensor_param_model_layers_90_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/mean":1.0560370355960913e-09,"train/train/tensor_act_model_layers_51_mlp_down_proj/std":0.015580831496054581,"train/train/tensor_act_model_layers_52_self_attn_k_proj/max_abs":1.0859375,"train/train/tensor_act_model_layers_48_mlp/std":0.015137602987144554,"train/train/tensor_act_model_layers_63_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_q_proj/std":0.22047577753177622,"train/train/layer_model_layers_15/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_o_proj_weight/max_abs":0.0888671875,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/norm":0.00010151509215644966,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_63_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_41_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/norm":0.03492747148904312,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47/mean":-0.0013132095336914062,"train/train/tensor_act_model_layers_90_self_attn_q_proj/std":0.2277946514070388,"train/train/tensor_act_model_layers_85_post_attention_layernorm/norm":5792.60327148473,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/max_abs":0.00135040283203125,"train/train/layer_model_layers_11/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_47_self_attn_v_proj/mean":0.0136871337890625,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp/max_abs":0.0849609375,"train/train/tensor_grad_model_layers_38_mlp_gate_proj_weight/std":9.420828080344963e-05,"train/train/layer__model_layers_75/param/norm":17.944957073646066,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_k_proj_weight/std":4.1592605724700654e-07,"train/train/tensor_param_model_layers_5_self_attn_k_proj_weight/max_abs":0.08837890625,"train/train/tensor_act_model_layers_4_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/mean":6.389617919921875e-05,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/max_abs":0.08544921875,"train/train/layer_model_layers_37/grad/std":0.0002670553628743747,"train/train/tensor_act_model_layers_77_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_mlp_gate_proj_weight/max_abs":0.08935546875,"train/train/tensor_act_model_layers_79_post_attention_layernorm/mean":0.00337982177734375,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/norm":0.00010294383063136013,"train/train/tensor_act_model_layers_77_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/mean":-6.46650732960552e-10,"train/train/tensor_act_model_layers_72_self_attn_k_proj/norm":1281.206478945007,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/std":0.0006091772806690696,"train/train/tensor_act_model_layers_62_mlp/mean":0.00022920966148376465,"train/train/tensor_act_model_layers_25/max_abs":1.171875,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/mean":0.0001220703125,"train/train/tensor_act_model_layers_14/max_abs":0.94921875,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/mean":5.5730342864990234e-06,"train/train/tensor_act_model_layers_77_self_attn_o_proj/std":0.05011241332546886,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_mlp_gate_proj_weight/norm":0.03847636111443967,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/norm":2.59375,"train/train/tensor_param_model_layers_79_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/std":0.00047109260665346455,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/norm":9.367999653652016e-05,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_73/act/max_abs":4.53125,"train/train/tensor_grad_model_layers_70_self_attn_k_proj_weight/mean":2.7212081477046013e-09,"train/train/tensor_act_model_layers_1_mlp_down_proj/mean":-0.0001379251480102539,"train/train/tensor_act_model_layers_56_self_attn_o_proj/norm":275.98644891460725,"train/train/tensor_param_model_layers_18_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/mean":-4.861503839492798e-06,"train/train/tensor_act_model_layers_54/norm":2102.476480125663,"train/train/tensor_grad_model_layers_74_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_mlp_up_proj_weight/norm":0.04802904496005287,"train/train/tensor_param_model_layers_7_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_43_self_attn_q_proj/max_abs":1.0859375,"train/train/tensor_param_model_layers_33_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_v_proj/max_abs":1.1953125,"train/train/tensor_act_model_layers_37_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_41_self_attn_q_proj/std":0.2255893364873179,"train/train/tensor_act_model_layers_75_mlp_gate_proj/mean":-0.00537872314453125,"train/train/tensor_param_model_layers_91_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/std":0.019775390625,"train/train/tensor_act_model_layers_18_mlp_down_proj/mean":0.00020074844360351562,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/max_abs":0.00131988525390625,"train/train/tensor_act_model_layers_78_input_layernorm/norm":5792.593994147318,"train/train/tensor_act_model_layers_10_post_attention_layernorm/mean":-0.06011962890625,"train/train/tensor_param_model_layers_58_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_57_self_attn_v_proj/norm":1302.0502519056452,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/max_abs":0.08203125,"train/train/tensor_grad_model_layers_10_mlp_up_proj_weight/norm":0.06719058717244282,"train/train/tensor_param_model_layers_29_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_54_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_24_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_64_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_post_attention_layernorm/mean":-0.0023784637451171875,"train/train/tensor_grad_model_layers_17_mlp_gate_proj_weight/std":0.0001691860931882327,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_v_proj/max_abs":1.015625,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_27/norm":1470.1112877599974,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/std":0.0198974609375,"train/train/layer__model_layers_34/param/mean":0.001522022550823908,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_mlp_gate_proj_weight/mean":3.261375240981579e-07,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/std":0.0001489937200732765,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/std":7.24783044639139e-05,"train/train/tensor_act_model_layers_25_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/mean":-7.343292236328125e-05,"train/train/tensor_act_model_layers_2_self_attn_k_proj/std":0.22753935831663646,"train/train/tensor_act_model_layers_67_self_attn_q_proj/std":0.2190014470889899,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/std":7.717882101516251e-05,"train/train/tensor_act_model_layers_82/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_48_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn_k_proj/std":0.23071796951556872,"train/train/tensor_act_model_layers_80_self_attn_k_proj/mean":0.0089874267578125,"train/train/tensor_act_model_layers_2_mlp_gate_proj/std":0.2224133952875158,"train/train/layer_model_layers_37/grad/mean":-1.1867643065181054e-07,"train/train/tensor_act_model_layers_83_mlp_up_proj/std":0.22876115913377615,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_3_input_layernorm/std":1.0000076315409225,"train/train/tensor_act_model_layers_89_self_attn/std":0.04742491269526409,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/mean":2.949673216789961e-08,"train/train/layer__model_layers_78/param/norm":17.935478707979193,"train/train/tensor_act_model_layers_42_post_attention_layernorm/std":1.0000043999074089,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/norm":0.03724685100071689,"train/train/tensor_grad_model_layers_40_mlp_up_proj_weight/mean":7.306225597858429e-07,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_input_layernorm/norm":5792.596801760718,"train/train/tensor_act_model_layers_0_mlp_up_proj/mean":-0.004547119140625,"train/train/tensor_act_model_layers_1_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_gate_proj/norm":1823.4401819209668,"train/train/tensor_act_model_layers_33_post_attention_layernorm/mean":-0.022186279296875,"train/train/layer__model_layers_48/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_mlp_gate_proj_weight/max_abs":0.00124359130859375,"train/train/tensor_param_model_layers_43_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_mlp_gate_proj_weight/max_abs":0.00848388671875,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/max_abs":0.0145263671875,"train/train/tensor_act_model_layers_20_self_attn_k_proj/mean":0.016754150390625,"train/train/tensor_act_model_layers_39_mlp_down_proj/mean":-0.0003447532653808594,"train/train/tensor_act_model_layers_69_self_attn_v_proj/norm":1327.9645186607802,"train/train/tensor_grad_model_layers_29_mlp_up_proj_weight/mean":-1.123175024986267e-06,"train/train/tensor_param_model_layers_32_self_attn_k_proj_weight/max_abs":0.08056640625,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/std":3.9899377504959596e-05,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/std":7.392563338306233e-05,"train/train/tensor_act_model_layers_72_mlp_down_proj/mean":0.00023245811462402344,"train/train/layer_model_layers_87/grad/norm":0.1299355033487424,"train/train/epoch_time_elapsed":80.14975232630968,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/max_abs":0.076171875,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn/norm":283.4490299270647,"train/train/tensor_act_model_layers_75_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/std":0.0004326172344594493,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/std":0.0010888675486907112,"train/train/tensor_act_model_layers_41_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_mlp_gate_proj_weight/norm":0.025795207435874108,"train/train/tensor_param_model_layers_1_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_mlp_down_proj/max_abs":0.0859375,"train/train/tensor_act_model_layers_21_input_layernorm/std":0.999031104154432,"train/train/layer_model_layers_37/act/norm":9062.216616581618,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/mean":-0.00010824203491210938,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/max_abs":0.0830078125,"train/train/tensor_grad_model_layers_0_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_gate_proj/norm":1862.8875478304737,"train/train/tensor_act_model_layers_11_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/norm":0.00027086239399664225,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_63_self_attn_q_proj/mean":0.00824737548828125,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/norm":0.0022199440558803994,"train/train/tensor_act_model_layers_3_self_attn/std":0.039127129065247804,"train/train/tensor_param_model_layers_43_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_57_self_attn_o_proj/max_abs":0.2119140625,"train/train/tensor_act_model_layers_15_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_15_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_5/act/std":0.4117119425115529,"train/train/tensor_grad_model_layers_18_self_attn_k_proj_weight/mean":-3.768946044147015e-09,"train/train/tensor_param_model_layers_36_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/std":5.980155009110105e-07,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_v_proj/max_abs":1.21875,"train/train/tensor_param_model_layers_10_mlp_up_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_14_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_92_input_layernorm_weight/mean":7.197260856628418e-06,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_mlp_gate_proj_weight/max_abs":0.091796875,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_44_post_attention_layernorm/mean":-0.00768280029296875,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/norm":0.12971658035923075,"train/train/tensor_act_model_layers_75_self_attn/max_abs":0.2197265625,"train/train/tensor_act_model_layers_5_self_attn_v_proj/norm":1294.4243961021011,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/mean":-2.0656734704971313e-06,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/mean":-6.100395694375038e-06,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_v_proj/norm":1276.2365125629503,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/mean":0.00015926361083984375,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_22/param/mean":0.0015128220485264724,"train/train/tensor_act_model_layers_37_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/std":0.0201416015625,"train/train/layer__model_layers_40/param/norm":17.936390699584045,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/max_abs":0.09423828125,"train/train/tensor_grad_model_layers_76_mlp_gate_proj_weight/max_abs":0.00125885009765625,"train/train/tensor_act_model_layers_28/norm":1511.784855987545,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/max_abs":0.0022430419921875,"train/train/tensor_grad_model_layers_85_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_self_attn_k_proj_weight/max_abs":3.9637088775634766e-06,"train/train/tensor_grad_model_layers_4_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/max_abs":9.894371032714844e-06,"train/train/tensor_act_model_layers_62_self_attn_q_proj/mean":-0.0010726451873779297,"train/train/tensor_param_model_layers_28_mlp_gate_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_3_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_mlp/mean":-0.00047469139099121094,"train/train/tensor_grad_model_layers_11_post_attention_layernorm_weight/max_abs":0.0004024505615234375,"train/train/tensor_param_model_layers_44_mlp_down_proj_weight/max_abs":0.08935546875,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/mean":-5.282461643218994e-06,"train/train/tensor_act_model_layers_22_self_attn_q_proj/max_abs":1.0234375,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/norm":0.11180334556296179,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_78/mean":0.0008978843688964844,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/norm":0.046366476360417734,"train/train/tensor_grad_model_layers_66_self_attn_o_proj_weight/mean":4.0605664253234863e-07,"train/train/tensor_act_model_layers_16_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_mlp_up_proj_weight/mean":5.657784640789032e-07,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/norm":0.15004065280618728,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_57_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_45/act/std":0.42031614314341786,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_68_self_attn/mean":0.0004940032958984375,"train/train/tensor_param_model_layers_58_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp_gate_proj/norm":1903.31457598587,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/max_abs":0.07861328125,"train/train/tensor_grad_model_layers_83_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_o_proj/max_abs":0.2119140625,"train/train/tensor_act_model_layers_38_self_attn_q_proj/mean":-0.00667572021484375,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_input_layernorm/mean":0.00865936279296875,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn_q_proj/max_abs":1.046875,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_25_self_attn_v_proj/norm":1253.4212229296647,"train/train/tensor_act_model_layers_6_self_attn_k_proj/norm":1265.563111026289,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_59_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_v_proj_weight/max_abs":0.007049560546875,"train/train/tensor_act_model_layers_53_mlp_gate_proj/max_abs":1.0546875,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_74_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_17_input_layernorm/norm":5792.534912113805,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_63_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/mean":9.322320693172514e-10,"train/train/tensor_act_model_layers_76_mlp_down_proj/mean":0.00033974647521972656,"train/train/tensor_act_model_layers_5_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/mean":-4.425644874572754e-06,"train/train/tensor_act_model_layers_48_mlp_up_proj/max_abs":1.296875,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_23_self_attn_o_proj_weight/norm":0.20348502311612116,"train/train/tensor_param_model_layers_31_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_self_attn_k_proj/max_abs":1.078125,"train/train/tensor_grad_model_layers_44_self_attn_k_proj_weight/norm":0.00013486591122481697,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/mean":-6.640329957008362e-07,"train/train/tensor_act_model_layers_13_mlp_gate_proj/max_abs":1.359375,"train/train/tensor_act_model_layers_67_self_attn_o_proj/std":0.04895110542683219,"train/train/tensor_param_model_layers_21_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_28_self_attn_v_proj/std":0.22998253199791774,"train/train/tensor_act_model_layers_41_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/mean":7.600647222716361e-09,"train/train/tensor_param_model_layers_81_self_attn_q_proj_weight/max_abs":0.0732421875,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/norm":0.0003367847884819082,"train/train/tensor_act_model_layers_76_self_attn/std":0.048590320733836546,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/mean":0.00020122528076171875,"train/train/layer_model_layers_31/grad/max_abs":0.00750732421875,"train/train/layer_model_layers_57/act/max_abs":4.9375,"train/train/tensor_act_model_layers_79_self_attn_q_proj/max_abs":1.265625,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/mean":-4.172325134277344e-05,"train/train/tensor_act_model_layers_71_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_input_layernorm/max_abs":4.5625,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_22_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_q_proj/mean":0.019927978515625,"train/train/tensor_param_model_layers_27_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_65_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_7_self_attn_q_proj/std":0.22681491387469074,"train/train/tensor_act_model_layers_36_self_attn_v_proj/std":0.22925309888014672,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/mean":-1.0535586625337601e-07,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/max_abs":0.083984375,"train/train/layer_model_layers_49/grad/std":0.00022038305191464662,"train/train/tensor_param_model_layers_53_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_self_attn_q_proj_weight/max_abs":0.07666015625,"train/train/tensor_act_model_layers_87_mlp/std":0.015411861504344237,"train/train/tensor_act_model_layers_18_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/mean":1.0669231414794922e-05,"train/train/layer__model_layers_27/param/max_abs":1,"train/train/tensor_act_model_layers_5_self_attn_q_proj/mean":0.016357421875,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/max_abs":0.001129150390625,"train/train/tensor_act_model_layers_17_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/mean":4.38690185546875e-05,"train/train/tensor_grad_model_layers_90_mlp_gate_proj_weight/norm":0.02462297699456092,"train/train/layer_model_layers_31/grad/std":0.0002869410487135171,"train/train/tensor_act_model_layers_40_input_layernorm/mean":-0.0173492431640625,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/max_abs":0.00152587890625,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_46_self_attn_q_proj/mean":-0.002559661865234375,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/mean":1.8153514247387648e-09,"train/train/tensor_act_model_layers_87_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp/norm":90.59024442844478,"train/train/tensor_act_model_layers_71_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/mean":-8.106231689453125e-05,"train/train/tensor_act_model_layers_52_input_layernorm/norm":5792.588989261726,"train/train/tensor_act_model_layers_65_self_attn_k_proj/mean":0.002613067626953125,"train/train/tensor_act_model_layers_62_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/norm":0.00010583341376629555,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/norm":0.00024194305786312266,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/norm":0.028180732079486726,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_77_self_attn_k_proj_weight/max_abs":5.245208740234375e-06,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/std":1.2354664672696878e-06,"train/train/tensor_param_model_embed_tokens_weight/norm":14.5,"train/train/tensor_act_model_layers_34/mean":-0.0057373046875,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/mean":-8.894858183339238e-09,"train/train/tensor_param_model_layers_89_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_self_attn_k_proj/mean":0.00498199462890625,"train/train/tensor_act_model_layers_30_self_attn_v_proj/mean":-0.002620697021484375,"train/train/tensor_act_model_layers_22_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_mlp_up_proj/mean":0.0011553764343261719,"train/train/tensor_act_model_layers_63_self_attn/max_abs":0.2197265625,"train/train/tensor_act_model_layers_66_self_attn_v_proj/mean":-0.006011962890625,"train/train/tensor_param_model_layers_27_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_18_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/mean":8.614733815193176e-07,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_69_input_layernorm/mean":0.0041027069091796875,"train/train/tensor_param_model_layers_65_mlp_down_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_40_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_k_proj/mean":0.004787445068359375,"train/train/tensor_grad_model_layers_66_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_self_attn_o_proj/std":0.04931716459432309,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_0_self_attn_o_proj_weight/max_abs":0.030029296875,"train/train/tensor_act_model_layers_43_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_54/act/max_abs":5.0625,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/max_abs":7.331371307373047e-06,"train/train/tensor_param_model_layers_60_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_8_post_attention_layernorm/std":1.0000044852394607,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/mean":-0.0001983642578125,"train/train/tensor_act_model_layers_77_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_54_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/max_abs":0.00083160400390625,"train/train/tensor_param_model_layers_0_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_4_self_attn_k_proj/max_abs":1.1875,"train/train/tensor_act_model_layers_3_mlp/max_abs":0.0859375,"train/train/tensor_act_model_layers_45_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_input_layernorm/mean":-0.0013275146484375,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/max_abs":0.09326171875,"train/train/tensor_act_model_layers_81_mlp_up_proj/mean":-0.0022640228271484375,"train/train/tensor_act_model_layers_90_self_attn_o_proj/std":0.052004333364874956,"train/train/tensor_act_model_layers_1_mlp_gate_proj/max_abs":1.2421875,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn/mean":0.002838134765625,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/std":0.0198974609375,"train/train/layer__model_layers_6/param/mean":0.0015734078917600063,"train/train/tensor_grad_model_layers_37_mlp_gate_proj_weight/mean":-2.337619662284851e-07,"train/train/tensor_param_model_layers_27_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_mlp_gate_proj_weight/std":0.0198974609375,"train/train/layer_model_layers_5/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_39_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_q_proj/max_abs":1.1640625,"train/train/tensor_act_model_layers_44_mlp_gate_proj/max_abs":1.109375,"train/train/tensor_act_model_layers_47_self_attn/norm":298.7526779173787,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_66_post_attention_layernorm/std":1.0000047073420484,"train/train/tensor_grad_model_layers_87_mlp_down_proj_weight/max_abs":0.000850677490234375,"train/train/tensor_param_model_layers_83_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/mean":4.144567355979234e-09,"train/train/tensor_act_model_layers_31_self_attn_k_proj/max_abs":1.015625,"train/train/tensor_act_model_layers_62_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_mlp_down_proj/std":0.015518339897702167,"train/train/tensor_act_model_layers_13_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_post_attention_layernorm/norm":5792.588623049061,"train/train/tensor_grad_model_layers_35_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/norm":0.00014721608819087182,"train/train/tensor_act_model_layers_88_self_attn_k_proj/norm":1319.6918140302755,"train/train/tensor_param_model_layers_14_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/mean":-2.0244624465703964e-07,"train/train/tensor_act_model_layers_92_self_attn_q_proj/std":0.22217354859167548,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_21_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_59_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_up_proj/max_abs":1.1875,"train/train/tensor_param_model_layers_11_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_43_self_attn_q_proj/mean":0.0082244873046875,"train/train/tensor_grad_model_layers_93_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_down_proj/std":0.015687010166326013,"train/train/tensor_act_model_layers_85_mlp_up_proj/mean":0.0036163330078125,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/max_abs":0.0047607421875,"train/train/tensor_param_model_layers_29_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/norm":0.20158743297523513,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/mean":2.6166439056396484e-05,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_88_post_attention_layernorm/mean":0.00983428955078125,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/max_abs":0.00124359130859375,"train/train/tensor_act_model_layers_80_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn/mean":-0.003543853759765625,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/max_abs":0.08251953125,"train/train/layer_model_layers_79/grad/mean":-1.188244924164274e-07,"train/train/tensor_act_model_layers_79_self_attn_q_proj/norm":1282.7422567735457,"train/train/tensor_param_model_layers_25_mlp_gate_proj_weight/mean":-2.872943878173828e-05,"train/train/tensor_act_model_layers_16_mlp_gate_proj/mean":-0.0013370513916015625,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/norm":0.03028205010741827,"train/train/tensor_grad_model_layers_3_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_mlp_up_proj/mean":0.0037994384765625,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_15/param/norm":17.93286498260596,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/std":0.0001163441242416577,"train/train/tensor_act_model_layers_80_post_attention_layernorm/norm":5792.591674812605,"train/train/tensor_act_model_layers_55_mlp_down_proj/mean":-0.0002148151397705078,"train/train/tensor_act_model_layers_90_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70/std":0.4184704142112513,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_mlp_up_proj/std":0.22363320500535128,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/max_abs":0.007598876953125,"train/train/tensor_act_model_layers_40_mlp_gate_proj/std":0.22412381347136331,"train/train/tensor_param_model_layers_38_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_post_attention_layernorm/max_abs":4.5625,"train/train/tensor_act_model_layers_0_mlp_gate_proj/max_abs":1.1796875,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/mean":9.679794311523438e-05,"train/train/tensor_grad_model_layers_89_post_attention_layernorm_weight/max_abs":0.0001468658447265625,"train/train/tensor_param_model_layers_75_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/mean":-4.523899406194687e-07,"train/train/tensor_act_model_layers_9_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_81/act/max_abs":4.78125,"train/train/tensor_act_model_layers_78_self_attn_o_proj/norm":296.95200205864137,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_36/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_49_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_91/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp_gate_proj/mean":-0.0076141357421875,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/std":0.00033482689053458226,"train/train/tensor_act_model_layers_40_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_self_attn_o_proj/std":0.04992779007769253,"train/train/layer_model_layers_40/grad/norm":0.1889683199042033,"train/train/tensor_grad_model_layers_56_mlp_gate_proj_weight/norm":0.028697559587549867,"train/train/tensor_act_model_layers_44_mlp_up_proj/norm":1876.778260436973,"train/train/tensor_param_model_layers_76_self_attn_o_proj_weight/mean":5.53131103515625e-05,"train/train/tensor_act_model_layers_69_mlp_gate_proj/std":0.2224148536579822,"train/train/tensor_act_model_layers_65_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_down_proj/norm":92.68890968088337,"train/train/tensor_act_model_layers_72_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_53_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_8_self_attn_v_proj/max_abs":1.15625,"train/train/tensor_param_model_layers_18_mlp_gate_proj_weight/mean":-0.000102996826171875,"train/train/tensor_act_model_layers_77_self_attn_q_proj/std":0.21900555348057518,"train/train/layer__model_layers_4/param/std":0.044257415176110086,"train/train/tensor_act_model_layers_27_post_attention_layernorm/max_abs":4.65625,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/norm":0.19175843515953603,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/std":9.538562172943276e-07,"train/train/tensor_param_model_layers_80_self_attn_k_proj_weight/max_abs":0.08349609375,"train/train/tensor_act_model_layers_47_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/mean":-3.2901763916015625e-05,"train/train/tensor_grad_model_layers_77_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_mlp_gate_proj_weight/mean":-6.48200511932373e-07,"train/train/tensor_grad_model_layers_67_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/mean":4.041939973831177e-06,"train/train/tensor_act_model_layers_32/std":0.28027874195166647,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/mean":1.8812716007232666e-07,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/norm":0.0006699655403220195,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_64_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_2/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/max_abs":0.004119873046875,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_input_layernorm/max_abs":5.34375,"train/train/tensor_param_model_layers_63_self_attn_v_proj_weight/max_abs":0.0771484375,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/mean":-0.0001087188720703125,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/std":4.6683782733646507e-07,"train/train/tensor_act_model_layers_34_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_mlp/mean":-8.702278137207031e-06,"train/train/tensor_act_model_layers_27_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/norm":9.603327528942878e-05,"train/train/tensor_act_model_layers_82_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_self_attn_k_proj/max_abs":1.078125,"train/train/tensor_act_model_layers_16_mlp_up_proj/max_abs":1.171875,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/mean":1.4479155652225018e-09,"train/train/tensor_act_model_layers_17_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn_o_proj/max_abs":0.2177734375,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/mean":2.699380274862051e-08,"train/train/tensor_act_model_layers_85_input_layernorm/max_abs":4.75,"train/train/tensor_param_model_layers_30_mlp_gate_proj_weight/mean":-2.574920654296875e-05,"train/train/tensor_act_model_layers_35_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/mean":-4.315376281738281e-05,"train/train/tensor_grad_model_layers_47_mlp_down_proj_weight/max_abs":0.0010833740234375,"train/train/tensor_act_model_layers_33_self_attn_o_proj/std":0.047241562047212374,"train/train/tensor_act_model_layers_61_mlp/norm":93.75454180365125,"train/train/tensor_act_model_layers_7_self_attn/norm":270.2847000196818,"train/train/tensor_param_model_layers_74_mlp_gate_proj_weight/max_abs":0.08544921875,"train/train/tensor_param_model_layers_56_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_87_mlp_down_proj/mean":-0.0002739429473876953,"train/train/tensor_grad_model_layers_47_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_53/param/mean":0.0015941268754265795,"train/train/tensor_act_model_layers_3_self_attn/max_abs":0.23046875,"train/train/tensor_act_model_layers_51_mlp_up_proj/mean":-0.002445220947265625,"train/train/tensor_act_model_layers_46_self_attn_v_proj/max_abs":1.078125,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/max_abs":0.08251953125,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/max_abs":0.09521484375,"train/train/tensor_act_model_layers_65_post_attention_layernorm/std":1.0000062398999894,"train/train/tensor_act_model_layers_4_mlp_up_proj/std":0.22949259610849895,"train/train/tensor_param_model_layers_24_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_20/act/norm":9021.264750808756,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_51_self_attn/std":0.049502662322945916,"train/train/tensor_act_model_layers_27_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_3_mlp_gate_proj/max_abs":1.21875,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_mlp_gate_proj_weight/std":0.00011729767007770164,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/mean":4.490138962864876e-07,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_77_mlp/norm":92.35353431537136,"train/train/tensor_param_model_layers_52_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_45_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_19/act/max_abs":4.5,"train/train/tensor_param_model_layers_0_self_attn_k_proj_weight/max_abs":0.0947265625,"train/train/tensor_grad_model_layers_2_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/max_abs":0.00128173828125,"train/train/tensor_param_model_layers_70_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_73/grad/std":0.00019455714533448087,"train/train/tensor_act_model_layers_33_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_72_self_attn/frac_near_dtype_limit":0,"train/train/layer_model_layers_76/grad/mean":-2.0792798063603001e-07,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_input_layernorm/std":0.999027336090625,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/norm":0.11802829477551027,"train/train/tensor_act_model_layers_5/mean":-0.006500244140625,"train/train/tensor_act_model_layers_70_input_layernorm/mean":-0.0019941329956054688,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/mean":-4.3335603550076485e-08,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/mean":-1.0371208190917969e-05,"train/train/tensor_grad_model_layers_31_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_self_attn_q_proj/mean":-0.007537841796875,"train/train/tensor_param_model_layers_69_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_63_mlp/max_abs":0.08154296875,"train/train/layer__model_layers_53/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_6_self_attn_o_proj/norm":258.17957838044504,"train/train/layer__model_layers_90/param/max_abs":1,"train/train/tensor_act_model_layers_30_mlp_down_proj/norm":96.91723356035952,"train/train/layer__model_layers_77/param/norm":17.9274593522932,"train/train/tensor_act_model_layers_51_mlp/max_abs":0.0849609375,"train/train/tensor_act_model_layers_40_self_attn_q_proj/std":0.2299853531152756,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_mlp_up_proj_weight/norm":0.10177677731968532,"train/train/tensor_act_model_layers_2_self_attn/norm":180.00567329164943,"train/train/tensor_param_model_layers_6_self_attn_q_proj_weight/max_abs":0.0703125,"train/train/tensor_grad_model_layers_25_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_v_proj_weight/max_abs":0.006103515625,"train/train/tensor_param_model_layers_79_self_attn_v_proj_weight/max_abs":0.07666015625,"train/train/tensor_param_model_layers_31_mlp_down_proj_weight/mean":-8.046627044677734e-06,"train/train/tensor_act_model_layers_15/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_30_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_18_post_attention_layernorm/norm":5792.538818362034,"train/train/tensor_act_model_layers_1_mlp/mean":-0.0001379251480102539,"train/train/tensor_grad_model_layers_2_self_attn_q_proj_weight/std":9.604052004002031e-06,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_72_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_88_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_14/act/mean":-0.008701767240251814,"train/train/tensor_act_model_layers_68_self_attn_v_proj/std":0.2292513557658411,"train/train/tensor_grad_model_layers_73_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp/mean":-0.00017890334129333496,"train/train/layer__model_layers_14/param/max_abs":1,"train/train/tensor_grad_model_layers_56_self_attn_q_proj_weight/max_abs":6.109476089477539e-06,"train/train/layer__model_layers_86/param/norm":17.932327215813345,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/mean":0.0001316070556640625,"train/train/tensor_param_model_layers_66_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_36_self_attn_v_proj/max_abs":1.0390625,"train/train/tensor_act_model_layers_15_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/mean":3.732202458195388e-08,"train/train/layer_model_layers_1/act/std":0.41038828242545267,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/std":0.0004094786702725723,"train/train/tensor_grad_model_layers_3_mlp_gate_proj_weight/std":0.00039192309391234857,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/mean":5.7697296142578125e-05,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/std":4.3206317096271915e-05,"train/train/tensor_param_model_layers_14_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_54_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/mean":-1.4469027519226074e-05,"train/train/tensor_act_model_layers_9_mlp/norm":88.50591759841689,"train/train/layer__model_layers_20/param/std":0.04422673002871929,"train/train/tensor_act_model_layers_73/mean":0.0007696151733398438,"train/train/tensor_param_model_layers_50_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_mlp_up_proj/max_abs":1.109375,"train/train/tensor_act_model_layers_58_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn_v_proj/mean":-0.003509521484375,"train/train/tensor_param_model_layers_81_mlp_up_proj_weight/norm":3.625,"train/train/layer_model_layers_83/act/std":0.4290809103588762,"train/train/tensor_act_model_layers_91_input_layernorm/std":1.000003910965089,"train/train/tensor_param_model_layers_70_mlp_up_proj_weight/mean":6.437301635742188e-05,"train/train/tensor_act_model_layers_13_self_attn_o_proj/mean":-0.0009279251098632812,"train/train/tensor_act_model_layers_84_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn_q_proj/norm":1321.8454863985219,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/max_abs":1.8596649169921875e-05,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/mean":0.0001239776611328125,"train/train/tensor_act_model_layers_92_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/std":5.014516669475064e-07,"train/train/tensor_param_model_layers_13_mlp_gate_proj_weight/max_abs":0.09326171875,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/max_abs":0.08251953125,"train/train/tensor_param_model_layers_37_self_attn_q_proj_weight/max_abs":0.09130859375,"train/train/tensor_act_model_layers_23_self_attn_q_proj/mean":0.002788543701171875,"train/train/tensor_act_model_layers_22_self_attn/mean":6.896257400512695e-05,"train/train/tensor_param_model_layers_15_self_attn_q_proj_weight/max_abs":0.07861328125,"train/train/tensor_param_model_layers_7_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/std":0.0009604179172935545,"train/train/tensor_grad_model_layers_4_mlp_gate_proj_weight/mean":5.820766091346741e-08,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_9_mlp_down_proj_weight/max_abs":0.08984375,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_36_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/norm":0.07230932615497683,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/norm":3.59375,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/max_abs":0.000843048095703125,"train/train/tensor_act_model_layers_21/norm":1311.095747755541,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_18_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_mlp/norm":91.16836865058079,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_q_proj/std":0.22754268684283957,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/mean":1.7182901501655579e-07,"train/train/tensor_act_model_layers_12_self_attn/norm":280.3868408330669,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/mean":9.30391252040863e-07,"train/train/tensor_grad_model_layers_89_self_attn_q_proj_weight/max_abs":4.380941390991211e-06,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn/mean":-0.0005165338516235352,"train/train/tensor_grad_model_layers_90_self_attn_v_proj_weight/norm":0.09487054637965113,"train/train/tensor_act_model_layers_21_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_12/act/max_abs":4.9375,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/mean":1.046457327902317e-06,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/max_abs":0.08740234375,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_64_self_attn_v_proj_weight/norm":0.10066767954369424,"train/train/tensor_act_model_layers_53_self_attn_o_proj/std":0.04638846666432787,"train/train/tensor_param_model_layers_74_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_27_mlp_gate_proj_weight/mean":5.005858838558197e-07,"train/train/layer_model_layers_47/grad/max_abs":0.005218505859375,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_92_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_16_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53/norm":2080.4401699687187,"train/train/tensor_act_model_layers_40_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_30_mlp_down_proj/std":0.01669377468626743,"train/train/tensor_act_model_layers_44_self_attn_v_proj/mean":-0.0048980712890625,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/norm":0.0028163318558072216,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/norm":0.7676824833638396,"train/train/tensor_grad_model_layers_65_mlp_down_proj_weight/max_abs":0.00130462646484375,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/std":0.0004042017827914745,"train/train/layer__model_layers_23/param/max_abs":1,"train/train/tensor_param_model_layers_83_mlp_up_proj_weight/mean":-0.00010061264038085938,"train/train/tensor_param_model_layers_76_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_30_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/mean":9.760260581970215e-07,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_58_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_51_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_51_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_mlp_gate_proj_weight/max_abs":0.083984375,"train/train/tensor_param_model_layers_28_mlp_down_proj_weight/mean":-0.00018978118896484375,"train/train/tensor_act_model_layers_72_self_attn_q_proj/std":0.2268122312023066,"train/train/tensor_grad_model_layers_20_mlp_gate_proj_weight/max_abs":0.001708984375,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_46_post_attention_layernorm/norm":5792.582275391535,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/max_abs":0.00160980224609375,"train/train/tensor_param_model_layers_77_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24/mean":-0.0112457275390625,"train/train/tensor_act_model_layers_67_input_layernorm/std":1.0000039875783464,"train/train/tensor_grad_model_layers_82_mlp_up_proj_weight/norm":0.022838406158242442,"train/train/layer_model_layers_9/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_54_self_attn_v_proj/max_abs":1.046875,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_52_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_input_layernorm/max_abs":4.5,"train/train/layer_model_layers_72/act/norm":9237.94117862086,"train/train/tensor_act_model_layers_50_input_layernorm/max_abs":4.625,"train/train/tensor_grad_model_layers_22_post_attention_layernorm_weight/std":6.709827756321758e-05,"train/train/tensor_act_model_layers_50_self_attn_k_proj/std":0.21607608225071545,"train/train/layer__model_layers_28/param/norm":17.94401830607487,"train/train/tensor_act_model_layers_38/std":0.3061578401615162,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/mean":5.024485290050507e-07,"train/train/tensor_param_model_layers_71_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_embed_tokens_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/norm":0.0001442816580692262,"train/train/layer__model_layers_85/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_mlp/max_abs":0.0830078125,"train/train/tensor_act_model_layers_35_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/mean":-9.202957153320312e-05,"train/train/tensor_act_model_layers_41_mlp_gate_proj/std":0.22046271032688977,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/mean":-3.0886440072208643e-09,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_60/param/norm":17.924320070786507,"train/train/tensor_act_model_layers_6_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/max_abs":4.202127456665039e-06,"train/train/tensor_act_model_layers_76_self_attn_o_proj/mean":0.0002543926239013672,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/max_abs":0.001129150390625,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_35_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_85_self_attn_k_proj_weight/max_abs":0.08154296875,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/mean":4.708766937255859e-06,"train/train/tensor_param_model_layers_60_self_attn_q_proj_weight/std":0.019775390625,"train/train/tensor_act_model_layers_82_self_attn_q_proj/std":0.22462108972828326,"train/train/tensor_grad_model_layers_77_mlp_gate_proj_weight/mean":-2.612359821796417e-07,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_input_layernorm/std":1.0000011101358928,"train/train/tensor_param_model_layers_14_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_1_mlp_gate_proj/mean":-0.00493621826171875,"train/train/tensor_param_model_layers_83_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_mlp/norm":91.6252518213992,"train/train/tensor_act_model_layers_41_mlp_gate_proj/max_abs":1.1953125,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/norm":0.17244971164271136,"train/train/tensor_act_model_layers_53_input_layernorm/frac_near_user_limit":0,"train/train/layer__model_layers_38/param/std":0.04425717603595561,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/std":3.592101359495602e-06,"train/train/layer_model_layers_34/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/std":0.00010711536971069719,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_60_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78_input_layernorm/mean":0.0034437179565429688,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/std":4.706580341691233e-07,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/max_abs":0.08251953125,"train/train/tensor_param_model_layers_4_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp_gate_proj/norm":1861.1795937966342,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/mean":-8.635222911834717e-06,"train/train/layer_model_layers_17/act/norm":8976.722004874326,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/max_abs":0.08837890625,"train/train/tensor_act_model_layers_67_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_v_proj_weight/std":0.0004605834549765543,"train/train/tensor_param_model_layers_77_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_40_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_mlp_gate_proj/mean":0.00778961181640625,"train/train/tensor_act_model_layers_54_self_attn_v_proj/norm":1285.6958822700005,"train/train/tensor_grad_model_layers_25_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp_down_proj/mean":-0.00043392181396484375,"train/train/tensor_act_model_layers_1_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_mlp_gate_proj_weight/mean":-4.959292709827423e-07,"train/train/layer_model_layers_0/grad/mean":-2.3837063269756514e-07,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_mlp_gate_proj_weight/mean":0.00017833709716796875,"train/train/tensor_act_model_layers_61_post_attention_layernorm/norm":5792.593750000587,"train/train/tensor_grad_model_layers_84_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_gate_proj_weight/max_abs":0.0810546875,"train/train/tensor_grad_model_layers_5_mlp_gate_proj_weight/std":0.0003182691853152511,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/norm":0.22035229911877188,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_67_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_38_mlp_down_proj_weight/std":9.8510047670422e-05,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/mean":-1.6361474990844727e-05,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_67_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_mlp_down_proj_weight/max_abs":0.08642578125,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_15/param/std":0.04425757929240996,"train/train/tensor_act_model_layers_87_mlp_down_proj/std":0.015411861504344237,"train/train/tensor_act_model_layers_31_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_mlp_gate_proj_weight/norm":0.06575047441318643,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_23_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_87/grad/mean":-1.9380935379151435e-07,"train/train/tensor_act_model_layers_78_post_attention_layernorm/max_abs":4.625,"train/train/tensor_act_model_layers_15_self_attn_k_proj/std":0.2321831590162695,"train/train/tensor_act_model_layers_9_self_attn_o_proj/std":0.046692791426449914,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/std":0.0001427409617932617,"train/train/tensor_grad_model_layers_53_mlp_gate_proj_weight/std":8.358105640003557e-05,"train/train/tensor_param_model_layers_2_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_63/max_abs":1.8984375,"train/train/tensor_act_model_layers_53_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_mlp_gate_proj_weight/norm":0.04612271320287745,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_83/act/mean":0.000747544424874442,"train/train/tensor_act_model_layers_48_mlp_gate_proj/mean":-0.003345489501953125,"train/train/tensor_grad_model_layers_17_mlp_up_proj_weight/max_abs":0.00153350830078125,"train/train/tensor_param_model_layers_71_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_self_attn_k_proj/norm":1371.9610946736152,"train/train/tensor_act_model_layers_26_self_attn_q_proj/mean":0.012298583984375,"train/train/tensor_grad_model_layers_48_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_71/act/norm":9261.656486276386,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_40_post_attention_layernorm/max_abs":4.875,"train/train/tensor_grad_model_layers_86_input_layernorm_weight/std":8.034260616134398e-05,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/norm":0.000892712516704852,"train/train/tensor_act_model_rotary_emb/max_abs":1,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/std":0.0196533203125,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/norm":0.0970089632199382,"train/train/tensor_param_model_layers_7_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_75_mlp_gate_proj_weight/mean":-0.000179290771484375,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_input_layernorm/norm":5792.579589846168,"train/train/tensor_act_model_layers_86_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_k_proj/max_abs":1.1015625,"train/train/tensor_param_model_layers_81_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_44_self_attn/std":0.049259785148007024,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/std":1.938262307407724e-06,"train/train/tensor_grad_model_layers_2_mlp_up_proj_weight/norm":0.18078411340712544,"train/train/tensor_grad_model_layers_51_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_mlp_gate_proj_weight/max_abs":0.08935546875,"train/train/tensor_act_model_layers_2_post_attention_layernorm/norm":5791.278198268138,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/max_abs":0.08251953125,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_11_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/norm":0.040565197283343714,"train/train/tensor_grad_model_layers_73_mlp_gate_proj_weight/max_abs":0.0009918212890625,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_90_input_layernorm_weight/mean":-2.7623027563095093e-06,"train/train/tensor_param_model_layers_31_self_attn_k_proj_weight/max_abs":0.08056640625,"train/train/layer__model_layers_49/param/mean":0.0015041095418230792,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/std":8.799107276588486e-05,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/max_abs":0.00013637542724609375,"train/train/layer_model_layers_35/act/std":0.41771960353693877,"train/train/tensor_act_model_layers_25/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_47_mlp_gate_proj/norm":1882.7285091146912,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn_o_proj/norm":284.9163664663061,"train/train/tensor_act_model_layers_68_self_attn/norm":294.7954106988249,"train/train/tensor_act_model_layers_80_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/max_abs":0.09521484375,"train/train/tensor_param_model_layers_80_self_attn_q_proj_weight/std":0.019775390625,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/max_abs":0.09130859375,"train/train/layer_model_layers_71/act/max_abs":4.71875,"train/train/tensor_param_model_layers_46_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_73_self_attn_v_proj/norm":1332.27307399996,"train/train/tensor_grad_model_layers_52_mlp_gate_proj_weight/std":8.812109817640753e-05,"train/train/tensor_act_model_layers_2_input_layernorm/max_abs":5.53125,"train/train/layer_model_layers_42/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_75_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_input_layernorm/max_abs":4.71875,"train/train/tensor_act_model_layers_83_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/max_abs":0.2099609375,"train/train/tensor_act_model_layers_52_self_attn/std":0.0494399258447518,"train/train/tensor_act_model_layers_19_post_attention_layernorm/std":1.0000079646389404,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_36/std":0.29639230498802954,"train/train/tensor_act_model_layers_5_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_11_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_38/act/norm":9077.362225016415,"train/train/tensor_param_model_layers_57_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_30/act/norm":9049.793012918162,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_76_self_attn_k_proj/mean":0.004497528076171875,"train/train/tensor_act_model_layers_4_mlp_up_proj/max_abs":1.203125,"train/train/tensor_grad_model_layers_16_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_59_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_59_mlp/norm":94.49744413156988,"train/train/tensor_act_model_layers_78_self_attn_o_proj/max_abs":0.263671875,"train/train/tensor_param_model_layers_1_mlp_gate_proj_weight/mean":0.000240325927734375,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/norm":0.0003795935268176803,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_post_attention_layernorm/mean":0.001819610595703125,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp/mean":-0.0007886886596679688,"train/train/tensor_grad_model_layers_72_input_layernorm_weight/norm":0.002130432434912598,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/max_abs":0.07666015625,"train/train/tensor_act_model_layers_64_mlp/std":0.015625380809680012,"train/train/tensor_grad_model_layers_12_self_attn_q_proj_weight/mean":1.2601958587765694e-08,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_15_post_attention_layernorm_weight/mean":-1.1041760444641113e-05,"train/train/tensor_act_model_layers_28_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_post_attention_layernorm_weight/std":4.256048231298948e-05,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_13/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_mlp_gate_proj_weight/norm":3.609375,"train/train/layer_model_layers_28/grad/max_abs":0.006317138671875,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_self_attn/max_abs":0.2119140625,"train/train/layer_model_layers_17/grad/frac_near_user_limit":0,"train/train/layer__model_layers_21/param/std":0.04422448769482385,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_59_self_attn/mean":-0.0011615753173828125,"train/train/tensor_grad_model_layers_17_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73/max_abs":2.078125,"train/train/tensor_act_model_layers_79_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_post_attention_layernorm/max_abs":4.96875,"train/train/tensor_param_model_layers_53_self_attn_o_proj_weight/max_abs":0.07373046875,"train/train/tensor_grad_model_layers_45_input_layernorm_weight/std":0.0001043666212442155,"train/train/layer_model_layers_20/act/frac_near_dtype_limit":0,"train/train/layer_model_layers_80/act/norm":9271.913013195002,"train/train/tensor_param_model_layers_28_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_self_attn_k_proj_weight/max_abs":0.08251953125,"train/train/tensor_param_model_layers_20_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_61_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn/std":0.018647465918248556,"train/train/tensor_act_model_layers_3_post_attention_layernorm/max_abs":5.03125,"train/train/tensor_param_model_layers_44_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_self_attn_k_proj/norm":1349.5612339699396,"train/train/layer_model_layers_29/grad/norm":0.21773527999226064,"train/train/tensor_param_model_layers_40_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/mean":-4.1443854570388794e-07,"train/train/tensor_act_model_layers_83_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/mean":-4.840694600716233e-08,"train/train/layer_model_layers_36/act/std":0.41983589208730804,"train/train/tensor_act_model_layers_61_self_attn_o_proj/max_abs":0.24609375,"train/train/tensor_act_model_layers_48_self_attn_q_proj/mean":-0.00775146484375,"train/train/tensor_param_model_layers_85_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp_gate_proj/std":0.2297397348757392,"train/train/tensor_param_model_layers_33_mlp_up_proj_weight/mean":-0.00013065338134765625,"train/train/tensor_act_model_layers_53_self_attn_k_proj/max_abs":1.1953125,"train/train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_self_attn_v_proj/max_abs":1.1484375,"train/train/tensor_act_model_layers_8_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_mlp_gate_proj_weight/max_abs":0.0830078125,"train/train/tensor_grad_model_layers_26_post_attention_layernorm_weight/std":5.63794132774346e-05,"train/train/tensor_grad_model_layers_79_input_layernorm_weight/mean":8.061528205871582e-06,"train/train/tensor_param_model_layers_45_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_26/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_v_proj/std":0.23072052229010928,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_41_self_attn_o_proj_weight/std":0.02001953125,"train/train/layer__model_layers_69/param/max_abs":1,"train/train/tensor_param_model_layers_48_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_89/std":0.4760851640390655,"train/train/tensor_grad_model_layers_72_mlp_gate_proj_weight/norm":0.026504020269003895,"train/train/tensor_param_model_layers_53_self_attn_k_proj_weight/mean":0.000308990478515625,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/max_abs":0.0025787353515625,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/mean":-7.009506225585938e-05,"train/train/tensor_act_model_layers_31_self_attn_o_proj/max_abs":0.2177734375,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/norm":0.000349038285575072,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/mean":5.372157829697244e-09,"train/train/layer__model_layers_92/param/norm":17.930999747730883,"train/train/tensor_grad_model_layers_39_input_layernorm_weight/mean":-7.1302056312561035e-06,"train/train/layer_model_layers_92/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/mean":3.918074071407318e-06,"train/train/tensor_act_model_layers_8_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_self_attn_v_proj_weight/max_abs":0.003997802734375,"train/train/tensor_act_model_layers_39_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn_k_proj/norm":1344.5223265789293,"train/train/tensor_act_model_layers_11/mean":-0.0098114013671875,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_self_attn_k_proj/mean":0.020263671875,"train/train/tensor_act_model_layers_27_post_attention_layernorm/norm":5792.559936525674,"train/train/tensor_grad_model_layers_71_mlp_up_proj_weight/mean":8.42846930027008e-08,"train/train/tensor_act_model_layers_76_mlp/norm":86.67957611304256,"train/train/tensor_act_model_layers_28_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_mlp_up_proj_weight/norm":0.046550581770067205,"train/train/tensor_grad_model_layers_21_self_attn_o_proj_weight/max_abs":0.007598876953125,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_mlp_gate_proj/max_abs":1.0234375,"train/train/tensor_act_model_layers_12_mlp_down_proj/norm":89.14210513432675,"train/train/layer_model_layers_66/grad/mean":2.048495127536159e-07,"train/train/tensor_act_model_layers_90_mlp_gate_proj/norm":1828.417716615885,"train/train/tensor_act_model_layers_61_self_attn_q_proj/std":0.22242154753712098,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/std":0.0003849598064990472,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_self_attn_v_proj/mean":-0.014129638671875,"train/train/tensor_param_model_layers_0_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_k_proj_weight/max_abs":4.32133674621582e-06,"train/train/tensor_act_model_layers_24_self_attn/norm":269.83750986738465,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/std":1.8856411167956882e-06,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/std":0.00017609825834691634,"train/train/tensor_param_model_layers_75_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/std":8.635641354009335e-05,"train/train/tensor_act_model_layers_38_self_attn/norm":295.6614942950424,"train/train/tensor_grad_model_layers_83_mlp_gate_proj_weight/mean":1.9371509552001953e-07,"train/train/layer__model_layers_19/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21/std":0.2258407363438487,"train/train/tensor_grad_model_layers_93_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/max_abs":0.09033203125,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/mean":3.8623809814453125e-05,"train/train/layer_model_layers_53/grad/std":0.00020013178853026906,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_49/mean":-0.0021419525146484375,"train/train/tensor_act_model_layers_44_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_self_attn_o_proj/norm":287.5787673178776,"train/train/tensor_grad_model_layers_17_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_85/act/std":0.42955929218463,"train/train/tensor_grad_model_layers_54_mlp_gate_proj_weight/max_abs":0.00145721435546875,"train/train/layer_model_layers_34/act/std":0.4180272933659132,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/max_abs":0.08837890625,"train/train/tensor_act_model_layers_5_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/norm":0.00013579565466463573,"train/train/tensor_act_model_layers_87_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_mlp_gate_proj/mean":-0.0011568069458007812,"train/train/tensor_grad_model_layers_32_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_k_proj_weight/mean":-4.793037078343332e-10,"train/train/tensor_act_model_layers_75_mlp_down_proj/mean":-0.00047397613525390625,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/mean":-2.08965502679348e-08,"train/train/tensor_grad_model_layers_68_mlp_gate_proj_weight/norm":0.029367362104842133,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_k_proj_weight/max_abs":6.139278411865234e-06,"train/train/tensor_grad_model_layers_40_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_mlp_up_proj_weight/mean":-2.2380845621228218e-07,"train/train/tensor_param_model_layers_38_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_post_attention_layernorm/std":0.9961049621549388,"train/train/layer_model_layers_84/act/max_abs":4.75,"train/train/tensor_act_model_layers_63_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn_o_proj/std":0.04596183097660219,"train/train/tensor_act_model_layers_58_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn/mean":0.0004239082336425781,"train/train/layer__model_layers_9/param/mean":0.0016126989760376549,"train/train/tensor_act_model_layers_1_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_mlp_gate_proj_weight/mean":-2.1047890186309814e-07,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/norm":3.609375,"train/train/layer__model_layers_79/param/mean":0.0014996059971183026,"train/train/tensor_act_model_layers_41_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_22_mlp_gate_proj_weight/std":0.0001331583215636336,"train/train/tensor_grad_model_layers_51_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_input_layernorm/std":1.0000032079354628,"train/train/tensor_act_model_layers_51_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_66/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/max_abs":0.006378173828125,"train/train/tensor_param_model_layers_1_mlp_gate_proj_weight/max_abs":0.07958984375,"train/train/tensor_act_model_layers_78_input_layernorm/std":1.000002102492608,"train/train/tensor_grad_model_layers_79_self_attn_v_proj_weight/std":0.00043110644499476456,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/max_abs":0.00014400482177734375,"train/train/tensor_act_model_layers_39_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_30_self_attn_k_proj/std":0.2211935114812981,"train/train/tensor_act_model_layers_83_self_attn_v_proj/norm":1313.8076433938463,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/norm":0.0001517286546392185,"train/train/tensor_grad_model_layers_62_input_layernorm_weight/mean":8.352100849151611e-06,"train/train/tensor_grad_model_layers_22_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/mean":-3.423338057473302e-09,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_23_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_70_self_attn_q_proj/std":0.22802876317441786,"train/train/tensor_grad_model_layers_21_mlp_gate_proj_weight/std":0.00013228799739037057,"train/train/tensor_grad_model_layers_31_mlp_up_proj_weight/std":0.00011177987263987007,"train/train/tensor_act_model_layers_0_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_param_model_norm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_14/param/mean":0.0015231770769854232,"train/train/tensor_act_model_layers_3/std":0.06531063932373857,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/std":3.667716988587865e-05,"train/train/tensor_grad_model_layers_57_mlp_down_proj_weight/max_abs":0.0015106201171875,"train/train/tensor_grad_model_layers_14_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_self_attn_o_proj/mean":-0.00014269351959228516,"train/train/tensor_act_model_layers_28_self_attn_q_proj/max_abs":1.0625,"train/train/tensor_act_model_layers_21_self_attn_q_proj/norm":1273.6237844846953,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_3_self_attn_q_proj_weight/std":5.909643902007675e-06,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/std":5.39658358007935e-05,"train/train/layer__model_layers_43/param/max_abs":1,"train/train/tensor_param_model_layers_66_self_attn_k_proj_weight/mean":-0.00010919570922851562,"train/train/tensor_act_model_layers_56_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_87/param/std":0.04424123131804849,"train/train/layer__model_layers_52/param/std":0.044264628420615376,"train/train/tensor_act_model_layers_27_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp_gate_proj/max_abs":1.1328125,"train/train/tensor_param_model_layers_60_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_input_layernorm/mean":-0.0994873046875,"train/train/tensor_act_model_layers_63_mlp/mean":0.00018638372421264648,"train/train/tensor_grad_model_layers_32_mlp_gate_proj_weight/norm":0.03871880968041551,"train/train/tensor_act_lm_head/std":0.2253422446762437,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/norm":2.53125,"train/train/tensor_act_model_layers_38_mlp/std":0.015518339897702167,"train/train/tensor_act_model_layers_76_mlp_down_proj/max_abs":0.08056640625,"train/train/tensor_grad_model_layers_43_mlp_up_proj_weight/mean":3.568129613995552e-07,"train/train/tensor_act_model_layers_51_mlp/mean":-0.0003571510314941406,"train/train/tensor_param_model_layers_27_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_24_mlp_gate_proj/std":0.22705252246804306,"train/train/tensor_grad_model_layers_59_self_attn_k_proj_weight/std":5.311517551185436e-07,"train/train/tensor_act_model_layers_14_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_87_mlp_down_proj_weight/mean":9.202957153320312e-05,"train/train/tensor_act_model_layers_88/norm":2745.4448018873272,"train/train/tensor_param_model_layers_24_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_27_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_89/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_37/param/std":0.04427820817017726,"train/train/tensor_param_model_layers_93_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_25_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/std":0,"train/train/tensor_act_/norm":16.642276130526163,"train/train/tensor_act_model_layers_0_mlp_up_proj/max_abs":1.1640625,"train/train/tensor_param_model_layers_60_mlp_gate_proj_weight/mean":-7.05718994140625e-05,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_51_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_29/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_93_post_attention_layernorm_weight/std":3.371001296857092e-05,"train/train/tensor_act_model_layers_26_self_attn_v_proj/std":0.22998563189694046,"train/train/tensor_param_model_layers_39_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_10_mlp_up_proj/std":0.22412414694142105,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/max_abs":0.078125,"train/train/tensor_param_model_layers_42_self_attn_k_proj_weight/max_abs":0.08349609375,"train/train/tensor_act_model_layers_25_mlp_up_proj/std":0.22315103301774433,"train/train/tensor_act_model_layers_77_self_attn_q_proj/mean":0.002539396286010742,"train/train/tensor_param_model_layers_14_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_88_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_91_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_input_layernorm/std":1.0000018647588953,"train/train/tensor_grad_model_layers_75_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_28/act/frac_near_user_limit":0,"train/train/layer__model_layers_87/param/mean":0.0015723218040049726,"train/train/tensor_param_model_layers_29_mlp_gate_proj_weight/max_abs":0.08447265625,"train/train/tensor_act_model_layers_81_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/max_abs":0.0003814697265625,"train/train/tensor_act_model_layers_70_mlp_up_proj/norm":1873.4777286500241,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_92/grad/mean":-1.709414162933129e-07,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_60_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28_self_attn_k_proj/norm":1332.5682717936909,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/max_abs":0.004852294921875,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/max_abs":0.000141143798828125,"train/train/tensor_act_model_layers_47_mlp_down_proj/norm":91.83005063890906,"train/train/tensor_param_model_layers_86_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_17_mlp_gate_proj_weight/max_abs":0.08837890625,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_mlp/mean":0.000789642333984375,"train/train/tensor_grad_model_layers_23_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_v_proj_weight/max_abs":0.080078125,"train/train/tensor_act_model_layers_70_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_48/param/std":0.044240931900947716,"train/train/tensor_act_model_layers_67_self_attn_k_proj/norm":1349.0072510464809,"train/train/tensor_act_model_layers_90_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_10_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_31_self_attn_o_proj/std":0.04742573627196074,"train/train/tensor_act_model_layers_85_input_layernorm/norm":5792.589599611027,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/max_abs":0.00099945068359375,"train/train/layer__model_layers_3/param/norm":17.922453946620983,"train/train/tensor_act_model_layers_85/std":0.4658305111554357,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/mean":-3.314018249511719e-05,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_mlp_up_proj_weight/norm":3.65625,"train/train/tensor_act_model_layers_16/norm":1123.9479270386878,"train/train/tensor_act_model_layers_58_self_attn_q_proj/mean":0.0125885009765625,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/mean":3.045424818992615e-07,"train/train/tensor_act_model_layers_61/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_93_self_attn/frac_near_dtype_limit":0,"train/train/layer__model_layers_80/param/max_abs":1,"train/train/tensor_param_model_layers_5_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/mean":-1.816079020500183e-06,"train/train/tensor_grad_model_layers_66_self_attn_v_proj_weight/norm":0.11081537321080434,"train/train/layer_model_layers_80/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_49_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_77/act/max_abs":4.71875,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/max_abs":5.900859832763672e-06,"train/train/tensor_act_model_layers_50_self_attn_v_proj/mean":0.0144195556640625,"train/train/tensor_act_model_layers_17_self_attn_v_proj/std":0.22925099416767655,"train/train/tensor_act_model_layers_56_mlp_down_proj/std":0.015335706115750525,"train/train/tensor_act_model_layers_1_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/max_abs":0.00011682510375976562,"train/train/tensor_act_model_layers_40_mlp/max_abs":0.08740234375,"train/train/layer_model_layers_13/grad/mean":-2.415481171688405e-06,"train/train/tensor_grad_model_layers_31_mlp_gate_proj_weight/norm":0.0435373656720683,"train/train/tensor_act_model_layers_7_self_attn_k_proj/std":0.22925151369359903,"train/train/tensor_grad_model_layers_32_mlp_gate_proj_weight/std":0.00010699578807641619,"train/train/tensor_act_model_layers_55_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_37_mlp_down_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_60_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/norm":8.99169997177851e-05,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/norm":0.0008551007896367325,"train/train/tensor_grad_model_layers_78_self_attn_k_proj_weight/mean":-5.184119800105691e-11,"train/train/tensor_param_model_layers_91_mlp_down_proj_weight/mean":-2.276897430419922e-05,"train/train/tensor_act_model_layers_81_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_self_attn/frac_near_dtype_limit":0,"train/train/layer__model_layers_34/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_post_attention_layernorm_weight/max_abs":0.00018024444580078125,"train/train/tensor_act_model_layers_71_post_attention_layernorm/std":1.0000013693251903,"train/train/tensor_act_model_layers_93_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/max_abs":0.000835418701171875,"train/train/tensor_param_model_layers_11_input_layernorm_weight/std":0,"train/train/layer_model_layers_22/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_85_mlp_down_proj_weight/std":7.405920273595827e-05,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/mean":-6.031990051269531e-05,"train/train/tensor_act_model_layers_48_self_attn_o_proj/std":0.05023454374586558,"train/train/tensor_param_model_layers_86_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_42_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_40_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_mlp_down_proj/max_abs":0.08203125,"train/train/tensor_grad_model_layers_66_mlp_gate_proj_weight/mean":1.4668330550193787e-08,"train/train/tensor_param_model_layers_22_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_2_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_51/max_abs":1.71875,"train/train/tensor_act_model_layers_26_post_attention_layernorm/max_abs":4.65625,"train/train/layer_model_layers_43/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn/norm":301.4854531791899,"train/train/tensor_act_model_layers_83_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp_down_proj/mean":-0.0003795623779296875,"train/train/tensor_act_model_layers_48_input_layernorm/norm":5792.580688476813,"train/train/tensor_grad_model_layers_59_mlp_gate_proj_weight/norm":0.030554725925906155,"train/train/tensor_act_model_layers_60_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn/max_abs":0.25390625,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_post_attention_layernorm/mean":-0.05853271484375,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_90_mlp_down_proj/norm":89.01714880258355,"train/train/tensor_grad_model_layers_92_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_input_layernorm/mean":0.0146331787109375,"train/train/tensor_param_model_layers_6_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp_gate_proj/norm":1873.6347186228027,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/mean":-6.388290785253048e-08,"train/train/tensor_act_model_layers_71_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer__model_layers_60/param/max_abs":1,"train/train/tensor_act_model_layers_65_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/max_abs":0.080078125,"train/train/tensor_act_model_layers_88_mlp_up_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_74/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/mean":-7.655471563339233e-07,"train/train/tensor_param_model_layers_67_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_19_self_attn_o_proj/mean":-0.0011930465698242188,"train/train/layer_model_layers_93/grad/std":0.00018444059226331827,"train/train/tensor_param_model_layers_89_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_39_mlp_gate_proj/mean":0.006134033203125,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/norm":0.10347559892804123,"train/train/tensor_param_model_layers_87_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_input_layernorm_weight/max_abs":0.00054168701171875,"train/train/tensor_act_model_layers_72_mlp_gate_proj/mean":-0.0010080337524414062,"train/train/tensor_act_model_layers_39_input_layernorm/norm":5792.57604980999,"train/train/tensor_act_model_layers_29_mlp/mean":-0.00024116039276123047,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_k_proj/norm":1322.8666272020484,"train/train/tensor_param_model_layers_29_mlp_gate_proj_weight/norm":3.59375,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_65_self_attn_o_proj/mean":0.00017781555652618408,"train/train/tensor_act_model_layers_57_input_layernorm/mean":-0.004253387451171875,"train/train/layer__model_layers_39/param/max_abs":1,"train/train/tensor_grad_model_layers_4_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_50_input_layernorm_weight/max_abs":0.000457763671875,"train/train/tensor_param_model_layers_36_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp_down_proj/std":0.01596221298288338,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_75/act/max_abs":4.75,"train/train/tensor_act_model_layers_43_post_attention_layernorm/max_abs":4.8125,"train/train/tensor_act_model_layers_83_post_attention_layernorm/max_abs":4.71875,"train/train/tensor_param_model_layers_10_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_53_mlp_down_proj_weight/max_abs":0.08203125,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_64_self_attn_v_proj/norm":1327.7990318418672,"train/train/tensor_grad_model_layers_70_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_post_attention_layernorm/std":1.0000019236449276,"train/train/tensor_grad_model_layers_36_mlp_gate_proj_weight/max_abs":0.00165557861328125,"train/train/tensor_grad_model_layers_27_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_mlp_up_proj_weight/mean":2.956390380859375e-05,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/std":4.953765254823406e-05,"train/train/layer_model_layers_21/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83_self_attn_q_proj/std":0.22241498765224077,"train/train/tensor_param_model_layers_82_self_attn_q_proj_weight/mean":-0.000125885009765625,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/std":0.0006016893364109424,"train/train/tensor_param_model_layers_1_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_12_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_89/grad/std":0.00015934887521356297,"train/train/tensor_param_model_layers_12_self_attn_v_proj_weight/std":0.020263671875,"train/train/layer__model_layers_17/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_v_proj_weight/max_abs":0.004486083984375,"train/train/tensor_act_model_layers_35_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_post_attention_layernorm/max_abs":4.9375,"train/train/tensor_grad_model_layers_64_mlp_up_proj_weight/mean":-1.7666025087237358e-07,"train/train/layer__model_layers_84/param/mean":0.0015991615616773108,"train/train/tensor_act_model_layers_39_self_attn_q_proj/norm":1281.3268336668782,"train/train/tensor_act_model_layers_88_self_attn_k_proj/std":0.22778860977069118,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/norm":0.00011596313034711228,"train/train/layer_model_layers_79/act/std":0.4278579866386557,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_51_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_2_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_post_attention_layernorm/max_abs":4.46875,"train/train/layer_model_layers_77/grad/max_abs":0.004608154296875,"train/train/tensor_param_model_layers_32_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn_o_proj/mean":0.0032196044921875,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/mean":-4.000961780548096e-06,"train/train/tensor_act_model_layers_3_self_attn_q_proj/norm":1320.9147818374704,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/mean":6.020854925736785e-10,"train/train/tensor_param_model_layers_0_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/norm":8.756066357497071e-05,"train/train/tensor_act_model_layers_70_self_attn_q_proj/max_abs":1.1796875,"train/train/tensor_param_model_layers_57_mlp_up_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_42_mlp_gate_proj_weight/mean":-5.4836273193359375e-05,"train/train/tensor_act_model_layers_57_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_self_attn_q_proj_weight/mean":-1.3142198440618813e-09,"train/train/tensor_act_model_layers_0_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/norm":0.09427843572546468,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_20_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/mean":5.145557224750519e-08,"train/train/tensor_act_model_layers_18_input_layernorm/std":1.0000020936109562,"train/train/tensor_act_model_layers_46_self_attn_k_proj/std":0.2214399145972058,"train/train/tensor_act_model_layers_63_post_attention_layernorm/mean":0.0003814697265625,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/max_abs":0.00154876708984375,"train/train/tensor_act_/max_abs":8.32306957244873,"train/train/tensor_act_model_layers_59_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_63_mlp_down_proj/norm":91.49716153139236,"train/train/tensor_act_model_layers_76_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_q_proj/std":0.22412669713338582,"train/train/tensor_act_model_layers_39_post_attention_layernorm/norm":5792.57763671979,"train/train/tensor_act_model_layers_3_mlp_gate_proj/std":0.22461023486014267,"train/train/tensor_act_model_layers_76_mlp_up_proj/norm":1806.014514794292,"train/train/layer_model_layers_56/act/norm":9143.111899197227,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_60_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_76_mlp_gate_proj/mean":0.003955841064453125,"train/train/tensor_param_model_layers_77_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/max_abs":0.07861328125,"train/train/tensor_grad_model_layers_8_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/max_abs":0.00122833251953125,"train/train/tensor_act_model_layers_17/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_mlp_up_proj_weight/mean":-4.947185516357422e-06,"train/train/tensor_grad_model_layers_42_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_39_mlp_gate_proj_weight/mean":6.437301635742188e-05,"train/train/tensor_grad_model_layers_56_self_attn_o_proj_weight/max_abs":0.003570556640625,"train/train/tensor_act_model_layers_43_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_9_mlp_down_proj/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_32_mlp_down_proj_weight/max_abs":0.0013275146484375,"train/train/tensor_grad_model_layers_46_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_35/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_mlp/mean":-0.0003380775451660156,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_87_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_mlp_gate_proj_weight/mean":6.738118827342987e-07,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/norm":0.00015309146398039973,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/max_abs":0.07861328125,"train/train/tensor_grad_model_layers_74_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_49/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_6_self_attn_o_proj_weight/max_abs":0.07666015625,"train/train/tensor_act_model_layers_62_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/mean":3.892637323588133e-09,"train/train/tensor_grad_model_layers_71_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_24/param/std":0.04425404323074985,"train/train/tensor_param_model_layers_19_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp_gate_proj/mean":-0.001903533935546875,"train/train/tensor_act_model_layers_19_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_87/act/norm":9311.477479309542,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_17_self_attn/mean":0.0007753372192382812,"train/train/layer_model_layers_89/act/std":0.4296464366854554,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/mean":5.82986103836447e-10,"train/train/tensor_act_model_layers_44_input_layernorm/max_abs":4.78125,"train/train/tensor_act_model_layers_87_self_attn/max_abs":0.255859375,"train/train/tensor_act_model_layers_86_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_self_attn_q_proj_weight/norm":0.00011223363102349194,"train/train/tensor_param_model_layers_81_mlp_gate_proj_weight/norm":3.625,"train/train/layer_model_layers_86/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/std":0.22705143352886803,"train/train/tensor_grad_model_layers_17_self_attn_o_proj_weight/max_abs":0.01007080078125,"train/train/tensor_grad_model_layers_66_mlp_gate_proj_weight/norm":0.02793230431308824,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/norm":0.00020979913778022821,"train/train/layer_model_layers_0/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/mean":-6.719346856698394e-09,"train/train/tensor_act_model_layers_16_mlp/mean":-0.0008697509765625,"train/train/tensor_grad_model_layers_59_mlp_gate_proj_weight/mean":-2.0302832126617432e-07,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/max_abs":0.0732421875,"train/train/tensor_act_model_layers_47_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp_gate_proj/std":0.2263219593833241,"train/train/tensor_grad_model_layers_0_mlp_down_proj_weight/std":0.0008700817219492697,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/std":4.6867469546671257e-07,"train/train/tensor_act_model_layers_0/mean":-0.0007410049438476562,"train/train/tensor_act_model_layers_9_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/norm":0.000764713608709795,"train/train/tensor_act_model_layers_18_self_attn_q_proj/norm":1280.9720365707803,"train/train/tensor_act_model_layers_59_mlp_down_proj/norm":94.49744413156988,"train/train/tensor_act_model_layers_70_self_attn_o_proj/std":0.048952948618697076,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/std":0.00010256005641401832,"train/train/tensor_grad_model_layers_3_mlp_up_proj_weight/std":0.0003529451679477061,"train/train/tensor_act_model_layers_77_mlp/max_abs":0.08203125,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/max_abs":0.0771484375,"train/train/tensor_act_model_layers_63_post_attention_layernorm/max_abs":4.6875,"train/train/tensor_act_/mean":8.32113790512085,"train/train/tensor_param_model_layers_5_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_v_proj/max_abs":1.125,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/norm":2.59375,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/max_abs":0.08642578125,"train/train/tensor_act_model_layers_10_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_mlp_up_proj_weight/mean":2.1350570023059845e-07,"train/train/tensor_act_model_layers_2_post_attention_layernorm/mean":-0.0054645538330078125,"train/train/tensor_param_model_layers_51_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_28_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_32_mlp_gate_proj_weight/mean":9.825453162193298e-07,"train/train/tensor_param_model_layers_87_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/std":0.00055224352848656,"train/train/layer__model_layers_30/param/norm":17.93601638028216,"train/train/tensor_act_model_layers_9/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/mean":9.250640869140625e-05,"train/train/tensor_act_model_layers_21_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_23/grad/mean":1.2062899233495843e-08,"train/train/tensor_act_model_layers_20_self_attn/max_abs":0.25390625,"train/train/tensor_act_model_layers_87_self_attn_v_proj/mean":0.002956390380859375,"train/train/tensor_grad_model_layers_40_self_attn_v_proj_weight/max_abs":0.0068359375,"train/train/tensor_param_model_layers_11_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_act_/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_self_attn_v_proj_weight/mean":-0.0001583099365234375,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/max_abs":0.0031890869140625,"train/train/tensor_param_model_layers_2_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_37_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_31/param/norm":17.93507714473233,"train/train/tensor_act_model_layers_79_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_up_proj/std":0.22632101874244764,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/max_abs":0.00017261505126953125,"train/train/layer_model_layers_76/grad/max_abs":0.004852294921875,"train/train/tensor_grad_model_layers_62_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_10_self_attn_q_proj/std":0.22388791812862952,"train/train/tensor_grad_model_layers_60_mlp_gate_proj_weight/std":7.979199344083487e-05,"train/train/tensor_act_model_layers_42_input_layernorm/mean":-0.010223388671875,"train/train/tensor_act_model_layers_65_mlp/mean":0.0003733634948730469,"train/train/tensor_grad_model_layers_2_mlp_down_proj_weight/std":0.00048792451990156346,"train/train/tensor_grad_model_layers_28_self_attn_v_proj_weight/std":0.0006381615843804658,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_mlp/max_abs":0.0830078125,"train/train/tensor_param_model_layers_18_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp_up_proj/mean":-0.013641357421875,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/std":0.00012550143147758313,"train/train/tensor_grad_model_layers_14_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_33_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_91_mlp/norm":90.9719906597224,"train/train/tensor_act_model_layers_15_mlp_down_proj/mean":-0.00047397613525390625,"train/train/tensor_act_model_layers_9_mlp_up_proj/std":0.222903355038705,"train/train/tensor_act_model_layers_77_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/norm":0.00022179694835112443,"train/train/layer_model_layers_34/grad/std":0.00028207531188714525,"train/train/tensor_act_model_layers_43_mlp/max_abs":0.07958984375,"train/train/tensor_act_model_layers_57_mlp_down_proj/max_abs":0.07958984375,"train/train/tensor_grad_model_layers_25_mlp_gate_proj_weight/max_abs":0.001251220703125,"train/train/tensor_grad_model_layers_52_mlp_down_proj_weight/max_abs":0.00128936767578125,"train/train/layer_model_layers_54/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/mean":6.883055903017521e-09,"train/train/layer__model_layers_50/param/mean":0.0015914369484935648,"train/train/tensor_param_model_layers_47_mlp_up_proj_weight/mean":8.96453857421875e-05,"train/train/tensor_grad_model_layers_76_self_attn_q_proj_weight/std":4.681122389382471e-07,"train/train/tensor_param_model_layers_32_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/mean":-8.153915405273438e-05,"train/train/tensor_param_model_layers_23_mlp_gate_proj_weight/max_abs":0.08251953125,"train/train/tensor_grad_model_layers_25_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_43/act/norm":9096.979839578808,"train/train/tensor_param_model_layers_50_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_15/std":0.18604289311618447,"train/train/tensor_grad_model_layers_86_mlp_down_proj_weight/norm":0.0257967237795043,"train/train/tensor_grad_model_layers_57_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/max_abs":0.08056640625,"train/train/tensor_grad_model_layers_87_self_attn_o_proj_weight/max_abs":0.00384521484375,"train/train/tensor_grad_model_layers_28_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_mlp_up_proj/mean":0.0009632110595703125,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/std":9.317951549228103e-06,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/max_abs":0.00823974609375,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_16_mlp_down_proj/mean":-0.0008697509765625,"train/train/layer_model_layers_24/grad/max_abs":0.005584716796875,"train/train/layer__model_layers_54/param/mean":0.0015813913806552262,"train/train/layer_model_layers_84/grad/std":0.0001663081914649335,"train/train/tensor_act_model_layers_88_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_mlp_gate_proj_weight/mean":-5.639158189296722e-07,"train/train/tensor_param_model_layers_15_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/mean":0.00018978118896484375,"train/train/tensor_param_model_layers_79_mlp_up_proj_weight/mean":-6.437301635742188e-05,"train/train/tensor_act_model_layers_3_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/mean":-4.33283275924623e-09,"train/train/tensor_act_model_layers_57/norm":2159.961411945023,"train/train/tensor_grad_model_layers_87_self_attn_q_proj_weight/norm":0.00010997585152137647,"train/train/tensor_act_model_layers_12_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_55_mlp_gate_proj/mean":-0.0032806396484375,"train/train/tensor_act_model_layers_29_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_mlp_gate_proj_weight/norm":0.033941254718945824,"train/train/tensor_act_model_rotary_emb/std":0.71875,"train/train/tensor_grad_model_layers_50_mlp_up_proj_weight/std":8.313920495669145e-05,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_mlp_gate_proj_weight/max_abs":0.001220703125,"train/train/tensor_param_model_layers_27_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_3_self_attn_o_proj/mean":-0.0012998580932617188,"train/train/tensor_grad_model_layers_63_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_86_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_mlp_up_proj/max_abs":1.1875,"train/train/tensor_act_model_layers_73_post_attention_layernorm/norm":5792.594848636387,"train/train/tensor_act_model_layers_63_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_86_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2/mean":-0.0004849433898925781,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_71_mlp_gate_proj/max_abs":1.09375,"train/train/tensor_act_model_layers_91_self_attn_o_proj/norm":272.3716872119266,"train/train/tensor_act_model_layers_85_mlp_down_proj/norm":92.75920276238986,"train/train/tensor_grad_model_layers_30_post_attention_layernorm_weight/norm":0.0012026767104543134,"train/train/tensor_act_model_layers_50_mlp_down_proj/max_abs":0.08544921875,"train/train/tensor_act_model_layers_57_post_attention_layernorm/mean":-0.001129150390625,"train/train/tensor_act_model_layers_50/norm":2023.3489879622189,"train/train/tensor_param_model_layers_49_mlp_gate_proj_weight/max_abs":0.08447265625,"train/train/tensor_act_model_layers_60_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/max_abs":0.08251953125,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_mlp_gate_proj_weight/std":6.726283983269804e-05,"train/train/tensor_param_model_layers_79_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/max_abs":0.0001354217529296875,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/std":0.020263671875,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/mean":4.100799560546875e-05,"train/train/tensor_act_model_layers_19_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_15_self_attn/norm":284.9163664663061,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_85_self_attn_q_proj/norm":1298.1831319078278,"train/train/tensor_act_model_layers_63_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_self_attn_q_proj_weight/max_abs":0.09716796875,"train/train/tensor_act_model_layers_66_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_29/mean":-0.010467529296875,"train/train/tensor_act_model_layers_6_mlp_gate_proj/mean":0.0008102655410766602,"train/train/tensor_grad_model_layers_72_mlp_gate_proj_weight/mean":2.898741513490677e-07,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/mean":6.023794412612915e-06,"train/train/tensor_param_model_layers_7_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/std":0.0005337413944047209,"train/train/tensor_param_model_layers_44_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/mean":-6.198883056640625e-05,"train/train/tensor_param_model_layers_46_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/std":0.0008441805264818558,"train/train/tensor_grad_model_layers_73_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_25/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_58/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_81_input_layernorm/frac_near_user_limit":0,"train/train/layer_model_layers_66/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_85_input_layernorm/mean":0.011322021484375,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/std":5.853618764619036e-07,"train/train/tensor_act_model_layers_66/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/mean":-0.000209808349609375,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/norm":0.03235807460969981,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/mean":1.4933903003111482e-09,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/norm":0.10622586368847996,"train/train/tensor_act_model_layers_63_mlp_gate_proj/norm":1870.7851490007372,"train/train/layer_model_layers_35/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_34_self_attn_o_proj/norm":281.18485254940197,"train/train/tensor_param_model_layers_62_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn/std":0.04925852194229224,"train/train/tensor_grad_model_layers_78_self_attn_o_proj_weight/mean":-5.722977221012115e-07,"train/train/tensor_grad_model_layers_60_self_attn_k_proj_weight/norm":0.00011628791345935158,"train/train/tensor_act_model_layers_30_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/max_abs":1.9431114196777344e-05,"train/train/tensor_act_model_layers_48_input_layernorm/mean":-0.00379180908203125,"train/train/tensor_act_model_layers_61_self_attn/std":0.04852484093583319,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_input_layernorm_weight/norm":0.0024187827943148435,"train/train/tensor_grad_model_layers_85_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_75_input_layernorm_weight/mean":-2.5387853384017944e-06,"train/train/tensor_grad_model_layers_36_post_attention_layernorm_weight/norm":0.001062535288431582,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/max_abs":0.07763671875,"train/train/tensor_act_model_layers_18_mlp_up_proj/std":0.2233903448620019,"train/train/tensor_act_model_layers_91_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_self_attn_q_proj_weight/std":1.3663749245679598e-06,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/mean":4.420144250616431e-10,"train/train/tensor_grad_model_layers_55_self_attn_q_proj_weight/norm":0.0001321965986497641,"train/train/tensor_act_model_layers_77/max_abs":2.15625,"train/train/tensor_param_model_layers_8_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/mean":5.7220458984375e-05,"train/train/tensor_grad_model_layers_83_mlp_gate_proj_weight/norm":0.02592126510574056,"train/train/tensor_act_model_layers_46_mlp_gate_proj/std":0.22827746286570086,"train/train/tensor_act_model_layers_6/mean":-0.0105743408203125,"train/train/tensor_grad_model_layers_15_mlp_down_proj_weight/std":0.00019945039367613173,"train/train/layer__model_layers_76/param/max_abs":1,"train/train/tensor_grad_model_layers_80_post_attention_layernorm_weight/max_abs":0.00012874603271484375,"train/train/tensor_act_model_layers_9_mlp_down_proj/norm":88.50591759841689,"train/train/layer__model_layers_16/param/norm":17.939541477905728,"train/train/tensor_grad_model_layers_46_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_92_mlp_gate_proj/max_abs":1.3671875,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_4/grad/frac_near_user_limit":0,"train/train/layer_model_layers_35/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_self_attn_k_proj_weight/mean":-0.00011301040649414062,"train/train/tensor_act_model_layers_39_input_layernorm/mean":-0.0176544189453125,"train/train/tensor_param_model_layers_76_input_layernorm_weight/mean":1,"train/train/layer_model_layers_90/act/std":0.43049405853148176,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_mlp_gate_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_39_self_attn_v_proj/std":0.22681441360474733,"train/train/tensor_grad_model_layers_88_self_attn_k_proj_weight/max_abs":4.202127456665039e-06,"train/train/layer_model_layers_71/grad/norm":0.15539769705039747,"train/train/tensor_grad_model_layers_67_post_attention_layernorm_weight/max_abs":0.00013065338134765625,"train/train/tensor_act_model_layers_25_self_attn/std":0.04553422607066122,"train/train/tensor_act_model_layers_28_self_attn_q_proj/norm":1294.5341632092334,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/max_abs":0.00408935546875,"train/train/tensor_param_model_layers_55_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/norm":0.0022720613599620685,"train/train/tensor_param_model_layers_31_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn/mean":-0.00030040740966796875,"train/train/tensor_grad_model_layers_34_post_attention_layernorm_weight/std":5.03021950360946e-05,"train/train/tensor_grad_model_layers_93_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_71_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_61_self_attn_o_proj/std":0.04852484093583319,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_87_input_layernorm/norm":5792.598510744074,"train/train/tensor_act_model_layers_25_mlp_gate_proj/max_abs":1.0703125,"train/train/tensor_act_model_layers_19_mlp_down_proj/std":0.015382231767609566,"train/train/tensor_act_model_layers_52_self_attn_o_proj/max_abs":0.2255859375,"train/train/tensor_param_model_layers_30_mlp_up_proj_weight/mean":3.886222839355469e-05,"train/train/tensor_grad_model_layers_22_mlp_gate_proj_weight/max_abs":0.00142669677734375,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/max_abs":0.08935546875,"train/train/tensor_act_model_layers_3_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_mlp/norm":89.14210513432675,"train/train/tensor_grad_model_layers_8_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/std":0.0004634044397724731,"train/train/tensor_act_model_layers_52_mlp_up_proj/norm":1851.2798978667981,"train/train/tensor_param_model_layers_39_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_v_proj/std":0.2253569642290902,"train/train/tensor_act_model_layers_21_self_attn_k_proj/norm":1311.1538269375271,"train/train/tensor_act_model_layers_14_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_2_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/norm":0.1068005714251201,"train/train/tensor_grad_model_layers_8_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_mlp/mean":0.0002675056457519531,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/max_abs":0.078125,"train/train/tensor_param_model_layers_88_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_50_self_attn_o_proj_weight/max_abs":0.083984375,"train/train/tensor_grad_model_layers_55_mlp_down_proj_weight/norm":0.028726946979903356,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/mean":9.103678166866302e-07,"train/train/tensor_act_model_layers_87_mlp_gate_proj/mean":0.005939483642578125,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_61_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_19_self_attn_k_proj/max_abs":1.15625,"train/train/tensor_param_model_layers_12_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_64_mlp_up_proj_weight/mean":6.389617919921875e-05,"train/train/tensor_param_model_layers_48_self_attn_q_proj_weight/max_abs":0.0927734375,"train/train/tensor_act_model_layers_31_mlp_down_proj/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_57_input_layernorm/norm":5792.5902099620425,"train/train/tensor_act_model_layers_4_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer_model_layers_57/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_q_proj/std":0.22168477418604948,"train/train/tensor_act_model_layers_31_self_attn/max_abs":0.2177734375,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/norm":0.02528682914066524,"train/train/tensor_param_model_layers_82_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_input_layernorm/norm":5790.309570335661,"train/train/tensor_param_model_layers_53_mlp_up_proj_weight/mean":-7.677078247070312e-05,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_69_mlp_down_proj/max_abs":0.08349609375,"train/train/tensor_act_model_layers_57_self_attn_k_proj/max_abs":1.09375,"train/train/tensor_act_model_layers_46_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_59_mlp_up_proj_weight/mean":-2.8133392333984375e-05,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/std":0.0004115345979567865,"train/train/tensor_act_model_layers_31_self_attn_k_proj/mean":4.6253204345703125e-05,"train/train/tensor_act_model_layers_0_mlp_down_proj/std":0.01556396670046014,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_51_mlp_down_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/std":9.063235494844955e-05,"train/train/tensor_grad_model_layers_69_self_attn_k_proj_weight/max_abs":4.231929779052734e-06,"train/train/tensor_act_model_layers_35_self_attn_k_proj/std":0.2268103927673041,"train/train/tensor_act_model_layers_48/norm":1971.7664501349489,"train/train/tensor_grad_model_layers_15_input_layernorm_weight/mean":-7.053837180137634e-06,"train/train/layer__model_layers_59/param/mean":0.001525495056057125,"train/train/tensor_act_model_layers_23_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_87_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp_gate_proj/std":0.22632130512410933,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_77_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45/norm":1915.1129431975717,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/max_abs":0.09716796875,"train/train/tensor_act_model_layers_54_input_layernorm/max_abs":4.96875,"train/train/tensor_act_model_layers_82_mlp/norm":88.89216252348889,"train/train/layer_model_layers_44/grad/std":0.0002535776048568315,"train/train/tensor_param_model_layers_90_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_34/grad/max_abs":0.007598876953125,"train/train/tensor_act_model_layers_54_mlp_down_proj/mean":0.0002741813659667969,"train/train/tensor_grad_model_layers_49_self_attn_o_proj_weight/max_abs":0.00543212890625,"train/train/tensor_act_model_layers_42_self_attn_v_proj/mean":-0.0065155029296875,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_77_mlp_up_proj_weight/std":7.073806361191756e-05,"train/train/tensor_act_model_layers_62_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_13_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn_v_proj/max_abs":1.1171875,"train/train/tensor_act_model_layers_55_mlp_up_proj/std":0.22535094442016643,"train/train/tensor_act_model_layers_39_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_31_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_post_attention_layernorm/std":1.0000149755548662,"train/train/tensor_act_model_layers_78_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_30/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_up_proj/norm":1859.7354657949136,"train/train/layer_model_layers_5/grad/norm":0.6564289274691256,"train/train/tensor_param_model_layers_37_mlp_up_proj_weight/mean":0.0003643035888671875,"train/train/tensor_act_model_layers_0_input_layernorm/max_abs":4.75,"train/train/layer__model_layers_49/param/max_abs":1,"train/train/tensor_grad_model_layers_56_input_layernorm_weight/max_abs":0.000396728515625,"train/train/tensor_act_model_layers_29_self_attn_v_proj/std":0.22558879975260196,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/std":8.195313854878329e-07,"train/train/layer_model_layers_92/act/std":0.43101663784007277,"train/train/tensor_act_model_layers_25_self_attn_q_proj/max_abs":1.0859375,"train/train/tensor_act_model_layers_82_self_attn_k_proj/mean":0.0011962652206420898,"train/train/tensor_param_model_layers_89_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_5/norm":541.6626193831363,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/norm":2.59375,"train/train/layer__model_layers_10/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_88/act/norm":9313.378171174687,"train/train/tensor_act_model_layers_59_mlp_gate_proj/norm":1857.2266780783973,"train/train/tensor_param_model_layers_44_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_mlp_up_proj/mean":-0.00440216064453125,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_1_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_grad_model_layers_75_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_67_self_attn_q_proj/norm":1268.7304614675008,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/std":3.1264197111524264e-05,"train/train/layer_model_layers_14/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_91_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_self_attn_o_proj/std":0.04937956841370703,"train/train/tensor_param_model_layers_36_self_attn_q_proj_weight/max_abs":0.08837890625,"train/train/tensor_act_model_layers_11_self_attn_q_proj/mean":-0.00762939453125,"train/train/tensor_act_model_layers_25_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/mean":-1.9315630197525024e-06,"train/train/layer_model_layers_3/act/std":0.4112590463501952,"train/train/tensor_act_model_layers_65_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_72_self_attn/std":0.050173352072587184,"train/train/layer_model_layers_85/grad/max_abs":0.004241943359375,"train/train/layer_model_layers_19/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_input_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_59_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_mlp_gate_proj_weight/max_abs":0.0021820068359375,"train/train/tensor_act_model_layers_63_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_mlp_up_proj/norm":1864.3357408422728,"train/train/tensor_act_model_layers_47_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_51_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn/norm":295.3285272532396,"train/train/tensor_act_model_layers_85_mlp_gate_proj/norm":1852.307428814954,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/mean":-2.6792287826538086e-05,"train/train/tensor_act_model_layers_22_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/max_abs":0.08447265625,"train/train/tensor_grad_model_layers_47_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/max_abs":0.0771484375,"train/train/tensor_grad_model_layers_76_self_attn_k_proj_weight/std":4.3991402460983335e-07,"train/train/tensor_grad_model_layers_36_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_48_mlp_up_proj_weight/max_abs":0.1064453125,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/mean":1.671724021434784e-07,"train/train/tensor_grad_model_layers_26_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/norm":0.09731569035763306,"train/train/tensor_param_model_layers_32_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_45_self_attn_k_proj_weight/max_abs":0.0830078125,"train/train/tensor_grad_model_layers_90_self_attn_q_proj_weight/std":4.837146430351089e-07,"train/train/tensor_act_model_layers_24/norm":1398.0477487669807,"train/train/tensor_grad_model_layers_13_mlp_gate_proj_weight/mean":1.8998980522155762e-06,"train/train/tensor_param_model_layers_77_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54_input_layernorm/std":1.0000104698371952,"train/train/layer_model_layers_70/grad/mean":1.9457286395213905e-07,"train/train/layer_model_layers_5/grad/mean":-8.209982622478756e-07,"train/train/tensor_act_model_layers_62_self_attn/std":0.05066094740115944,"train/train/tensor_act_model_layers_55_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_post_attention_layernorm_weight/std":3.182167586064655e-05,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/std":0.00010613351734649292,"train/train/tensor_param_model_layers_80_mlp_down_proj_weight/mean":9.584426879882812e-05,"train/train/layer_model_layers_50/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_o_proj/std":0.00869760555080741,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_76/act/max_abs":4.71875,"train/train/tensor_param_model_layers_75_self_attn_q_proj_weight/mean":-4.482269287109375e-05,"train/train/layer_model_layers_62/act/max_abs":4.8125,"train/train/tensor_act_model_layers_20_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_post_attention_layernorm/std":1.0000024344246794,"train/train/tensor_act_model_layers_93_post_attention_layernorm/mean":0.01947021484375,"train/train/tensor_act_model_layers_62/max_abs":1.9140625,"train/train/tensor_act_model_layers_28_mlp_gate_proj/max_abs":1.1640625,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_29_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_27_self_attn_v_proj/norm":1276.9288558494484,"train/train/tensor_param_model_layers_16_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_14_post_attention_layernorm/norm":5792.517700196294,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_28_mlp_gate_proj_weight/mean":9.08970832824707e-07,"train/train/tensor_param_model_layers_83_self_attn_o_proj_weight/max_abs":0.0810546875,"train/train/tensor_grad_model_layers_34_input_layernorm_weight/norm":0.0026485429654179104,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_68/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_q_proj/norm":1309.8757467453809,"train/train/tensor_act_model_layers_19_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/norm":0.23597324934175742,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/norm":0.00011912056482146893,"train/train/tensor_act_model_layers_71_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_58_self_attn_q_proj_weight/mean":2.6284396881237626e-10,"train/train/tensor_act_model_layers_66_self_attn_v_proj/std":0.22973998367855122,"train/train/tensor_act_model_layers_30_self_attn/max_abs":0.2158203125,"train/train/tensor_act_model_layers_10_self_attn_k_proj/mean":0.00621795654296875,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/mean":0.00019073486328125,"train/train/tensor_grad_model_layers_13_self_attn_o_proj_weight/mean":-1.1324882507324219e-05,"train/train/tensor_grad_model_layers_1_self_attn_q_proj_weight/mean":-4.311732482165098e-08,"train/train/tensor_act_model_layers_14_self_attn_o_proj/max_abs":0.2451171875,"train/train/tensor_grad_model_layers_68_input_layernorm_weight/std":7.626537917069444e-05,"train/train/tensor_act_model_layers_32/norm":1623.297590039074,"train/train/tensor_act_model_layers_48_self_attn_q_proj/std":0.21924342232264563,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_input_layernorm_weight/mean":-1.0877847671508789e-05,"train/train/tensor_grad_model_layers_48_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_20_self_attn_o_proj/mean":-0.000804901123046875,"train/train/tensor_act_model_layers_25_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_59_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_74_mlp_down_proj_weight/norm":0.029314417355424283,"train/train/tensor_act_model_layers_0_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer__model_layers_67/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_self_attn_v_proj/norm":1326.69285085482,"train/train/tensor_param_model_layers_80_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_29_self_attn_o_proj/norm":277.2471470731721,"train/train/tensor_grad_model_layers_70_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_self_attn_o_proj/std":0.049561444386567476,"train/train/tensor_act_model_layers_88_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_k_proj_weight/max_abs":3.039836883544922e-05,"train/train/layer_model_layers_27/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_41_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_78/norm":2574.366867882168,"train/train/layer__model_layers_2/param/norm":17.943106702157237,"train/train/tensor_param_model_layers_7_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_52/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp/std":0.015671156102791833,"train/train/layer_model_layers_65/grad/std":0.00021087503337963157,"train/train/tensor_param_model_layers_50_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/mean":-4.0618033381178975e-09,"train/train/tensor_param_model_layers_9_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_69_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/norm":0.41737623196146145,"train/train/tensor_grad_model_layers_68_self_attn_q_proj_weight/norm":0.00011997975278832985,"train/train/tensor_act_model_layers_10_self_attn/std":0.04596183097660219,"train/train/layer_model_layers_60/grad/std":0.0001979017023775628,"train/train/tensor_grad_model_layers_66_mlp_gate_proj_weight/std":7.715608127731493e-05,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/norm":0.025443315853387095,"train/train/layer_model_layers_80/act/std":0.42777582009900544,"train/train/tensor_grad_model_layers_18_mlp_up_proj_weight/norm":0.053035582294088855,"train/train/tensor_act_model_layers_4_mlp_gate_proj/max_abs":1.1875,"train/train/tensor_param_model_layers_10_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_post_attention_layernorm/max_abs":4.59375,"train/train/tensor_act_model_layers_45_self_attn_k_proj/std":0.2221714029887555,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_88/act/max_abs":4.84375,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_76_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/mean":0.000156402587890625,"train/train/tensor_act_model_layers_56_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_mlp_up_proj_weight/mean":6.29425048828125e-05,"train/train/tensor_act_model_layers_34_self_attn_k_proj/max_abs":1.34375,"train/train/tensor_act_model_layers_77_self_attn_o_proj/mean":-0.0005682706832885742,"train/train/tensor_grad_model_layers_30_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_13/param/norm":17.940466871363743,"train/train/tensor_grad_model_layers_13_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_69_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_42_post_attention_layernorm/mean":-0.0100250244140625,"train/train/tensor_act_model_layers_5_post_attention_layernorm/norm":5792.256225586888,"train/train/tensor_grad_model_layers_82_post_attention_layernorm_weight/max_abs":0.0001544952392578125,"train/train/tensor_grad_model_layers_69_post_attention_layernorm_weight/max_abs":0.0001201629638671875,"train/train/tensor_act_model_layers_8_self_attn_o_proj/max_abs":0.25,"train/train/tensor_grad_model_layers_54_post_attention_layernorm_weight/max_abs":0.00015926361083984375,"train/train/tensor_param_model_layers_39_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_84_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65/norm":2327.6427799410653,"train/train/tensor_act_model_layers_53_mlp/mean":1.8224120140075684e-05,"train/train/tensor_grad_model_layers_33_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_4/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_41_input_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_60/act/norm":9171.480784632835,"train/train/tensor_act_model_layers_43_self_attn_k_proj/mean":0.000453948974609375,"train/train/tensor_act_model_layers_83_mlp_down_proj/mean":0.000812530517578125,"train/train/layer_model_layers_39/act/std":0.41938806923436645,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/max_abs":0.0025787353515625,"train/train/tensor_grad_model_layers_72_self_attn_v_proj_weight/max_abs":0.003814697265625,"train/train/layer_model_layers_42/grad/norm":0.1921685559356137,"train/train/tensor_act_model_layers_29_mlp/max_abs":0.083984375,"train/train/tensor_act_model_layers_80_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_86/param/mean":0.0015694563176814182,"train/train/tensor_act_model_layers_55_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn_o_proj/max_abs":0.2333984375,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/max_abs":1,"train/train/layer_model_layers_40/grad/max_abs":0.0068359375,"train/train/tensor_act_model_layers_39_self_attn_v_proj/mean":-0.01513671875,"train/train/tensor_act_model_layers_58_mlp_down_proj/norm":91.6252518213992,"train/train/tensor_act_model_layers_67_mlp_down_proj/norm":90.82348488364937,"train/train/tensor_param_model_layers_50_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_16_mlp_gate_proj_weight/mean":-7.915496826171875e-05,"train/train/tensor_grad_model_layers_49_mlp_gate_proj_weight/max_abs":0.0009918212890625,"train/train/tensor_act_model_layers_88_self_attn_o_proj/max_abs":0.2236328125,"train/train/tensor_grad_model_layers_24_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp/frac_near_dtype_limit":0,"train/train/layer_model_layers_10/grad/max_abs":0.0113525390625,"train/train/tensor_act_model_layers_75_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_53_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_mlp_up_proj_weight/norm":0.031684948052870435,"train/train/tensor_param_model_layers_57_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_75_mlp_up_proj/max_abs":1.1015625,"train/train/tensor_act_model_layers_56/norm":2133.7399948308544,"train/train/tensor_act_model_layers_47_post_attention_layernorm/mean":-0.0039196014404296875,"train/train/tensor_act_model_layers_13_self_attn_q_proj/max_abs":1.078125,"train/train/tensor_act_model_layers_61_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_input_layernorm_weight/mean":-3.644824028015137e-05,"train/train/tensor_act_model_layers_78_mlp_up_proj/std":0.22680842169974946,"train/train/tensor_param_model_layers_74_self_attn_k_proj_weight/max_abs":0.07421875,"train/train/tensor_act_model_layers_63_mlp_up_proj/max_abs":1.21875,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/max_abs":0.0791015625,"train/train/tensor_grad_model_layers_19_self_attn_o_proj_weight/std":0.0008607699166872971,"train/train/tensor_act_model/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/std":4.334886003728886e-07,"train/train/tensor_act_model_layers_53_mlp_down_proj/mean":1.8224120140075684e-05,"train/train/tensor_param_model_layers_27_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_41_self_attn_q_proj/mean":-0.0045013427734375,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/max_abs":0.0016937255859375,"train/train/tensor_act_model_layers_71_mlp/mean":0.000637054443359375,"train/train/tensor_grad_model_layers_36_self_attn_v_proj_weight/max_abs":0.007110595703125,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_24_mlp_up_proj_weight/mean":-1.7881393432617188e-05,"train/train/tensor_act_model_layers_32_self_attn_o_proj/norm":273.99505625892175,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_3_post_attention_layernorm_weight/max_abs":0.00089263916015625,"train/train/layer__model_layers_37/param/max_abs":1,"train/train/tensor_act_model_layers_58_mlp_up_proj/norm":1849.1342690408492,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_69_mlp_down_proj/norm":89.58783643347356,"train/train/tensor_grad_model_layers_89_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_50_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_32_self_attn_v_proj/std":0.2197378721240579,"train/train/tensor_grad_model_layers_51_post_attention_layernorm_weight/std":3.8530726034268545e-05,"train/train/layer__model_layers_88/param/max_abs":1,"train/train/tensor_grad_model_layers_28_self_attn_o_proj_weight/max_abs":0.006317138671875,"train/train/tensor_grad_model_layers_16_mlp_up_proj_weight/norm":0.06797551965434197,"train/train/tensor_act_model_layers_19_self_attn_q_proj/mean":0.003192901611328125,"train/train/tensor_param_model_layers_64_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/norm":9.091254009090466e-05,"train/train/tensor_param_model_layers_45_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp/norm":92.35709443455691,"train/train/tensor_grad_model_layers_91_self_attn_q_proj_weight/mean":-6.539266905747354e-10,"train/train/tensor_grad_model_layers_24_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/max_abs":0.01025390625,"train/train/tensor_param_model_layers_69_mlp_gate_proj_weight/mean":-4.291534423828125e-05,"train/train/tensor_act_model_layers_60_self_attn_v_proj/max_abs":1.1015625,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/max_abs":0.0009765625,"train/train/layer_model_layers_71/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_78_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_k_proj/norm":1318.047639920341,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/max_abs":0.00016498565673828125,"train/train/tensor_act_model_layers_74_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/std":4.7539114794177765e-05,"train/train/tensor_param_model_layers_48_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_12_mlp_gate_proj/std":0.22876105516817313,"train/train/tensor_act_model_layers_65_mlp_down_proj/norm":94.09292112202088,"train/train/tensor_param_model_layers_93_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_52_mlp_gate_proj/norm":1864.0014878607667,"train/train/tensor_act_model_layers_53_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp_up_proj/norm":1848.2442941638706,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_input_layernorm/std":0.9990308710979537,"train/train/tensor_grad_model_layers_9_mlp_gate_proj_weight/mean":9.224459063261747e-08,"train/train/tensor_act_model_layers_14_mlp_down_proj/max_abs":0.0830078125,"train/train/tensor_act_model_layers_91_self_attn_v_proj/max_abs":1.046875,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_88_input_layernorm_weight/max_abs":0.000659942626953125,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_72/grad/norm":0.15022676918423097,"train/train/tensor_act_model_layers_77_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_83/grad/norm":0.13679179118323234,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/norm":0.0008568857932276158,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/mean":-7.486343383789062e-05,"train/train/tensor_act_model_layers_71_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/mean":3.504753112792969e-05,"train/train/tensor_act_model_layers_12_self_attn_k_proj/mean":-0.010406494140625,"train/train/tensor_act_model_layers_89_post_attention_layernorm/norm":5792.602661138507,"train/train/tensor_act_model_layers_70_mlp_down_proj/mean":0.00025397539138793945,"train/train/tensor_grad_model_layers_56_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/norm":2.53125,"train/train/tensor_grad_model_layers_7_post_attention_layernorm_weight/std":0.00011797394808221052,"train/train/tensor_act_model_layers_66_mlp_gate_proj/std":0.22949782629502782,"train/train/tensor_param_model_layers_15_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_input_layernorm/std":1.0000011995427567,"train/train/tensor_act_model_layers_65_self_attn_q_proj/mean":0.0155792236328125,"train/train/tensor_act_model_layers_62_self_attn_q_proj/std":0.23047747895088164,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/mean":4.777684807777405e-07,"train/train/tensor_param_model_layers_25_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/norm":0.11292922472124999,"train/train/tensor_act_model_layers_58_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_k_proj/mean":-0.010528564453125,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/std":0.02001953125,"train/train/layer_model_layers_89/act/max_abs":4.625,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_85_self_attn_v_proj/mean":-0.00713348388671875,"train/train/tensor_act_model_layers_61_self_attn_v_proj/norm":1317.4899883743694,"train/train/tensor_grad_model_layers_74_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_up_proj/norm":1828.6944949951433,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_23_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_83_post_attention_layernorm/std":1.0000007069899994,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_74_post_attention_layernorm/std":1.0000007763183418,"train/train/tensor_act_model_layers_58_self_attn_o_proj/mean":-0.0012836456298828125,"train/train/tensor_param_model_layers_41_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_64_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_11_mlp_gate_proj_weight/norm":0.07361720397813012,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_self_attn_v_proj/mean":-0.0213623046875,"train/train/tensor_act_model_layers_60_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_65_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_self_attn_q_proj_weight/std":4.4784646314767327e-07,"train/train/tensor_param_model_layers_71_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn_q_proj/max_abs":1.0390625,"train/train/tensor_param_model_layers_66_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/max_abs":0.001434326171875,"train/train/tensor_param_model_layers_64_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/std":4.117674528000213e-07,"train/train/layer_model_layers_62/act/std":0.4242921701649859,"train/train/tensor_act_model_layers_87_self_attn_q_proj/norm":1294.256324022088,"train/train/tensor_grad_model_layers_70_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_37_self_attn_k_proj/std":0.21899883066958045,"train/train/tensor_act_model_layers_59_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp/norm":88.10078045245818,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_17_mlp_up_proj/norm":1890.7795145247244,"train/train/tensor_act_model_layers_41_self_attn_o_proj/frac_near_user_limit":0,"train/train/layer_model_layers_85/act/max_abs":4.75,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/mean":1.5888363122940063e-06,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_2_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_31_mlp/norm":91.61138897854775,"train/train/tensor_param_model_layers_38_self_attn_v_proj_weight/mean":0.00010204315185546875,"train/train/tensor_act_model_layers_8_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_mlp_gate_proj/mean":0.00273895263671875,"train/train/tensor_param_model_layers_56_self_attn_q_proj_weight/mean":0.00014400482177734375,"train/train/layer_model_layers_82/act/norm":9293.087048527312,"train/train/tensor_param_model_layers_18_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/mean":-4.5192427933216095e-07,"train/train/tensor_grad_model_layers_70_post_attention_layernorm_weight/std":3.393459726459e-05,"train/train/tensor_grad_model_layers_73_self_attn_v_proj_weight/norm":0.10425262083788879,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/std":4.893323875079995e-07,"train/train/tensor_act_model_layers_10/frac_near_dtype_limit":0,"train/train/layer__model_layers_48/param/max_abs":1,"train/train/tensor_param_model_layers_69_post_attention_layernorm_weight/std":0,"train/train/layer_model_layers_74/grad/std":0.0001994540773406524,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_91_mlp_up_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_mlp_down_proj/std":0.015443274157754145,"train/train/tensor_act_model_layers_51_self_attn_o_proj/std":0.049502662322945916,"train/train/tensor_grad_model_layers_8_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/std":6.264296264356442e-07,"train/train/tensor_act_model_layers_41_mlp_down_proj/mean":-0.000316619873046875,"train/train/layer__model_layers_25/param/max_abs":1,"train/train/tensor_param_model_layers_89_mlp_up_proj_weight/norm":3.640625,"train/train/layer_model_layers_77/grad/std":0.00019421496035747134,"train/train/tensor_act_model_layers_77_mlp_up_proj/norm":1875.8378216594608,"train/train/layer_model_layers_68/act/mean":-0.0018197298049926758,"train/train/tensor_act_model_layers_42_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_self_attn_o_proj_weight/mean":-2.9802322387695312e-05,"train/train/tensor_act_model_layers_38_self_attn_k_proj/mean":-0.018035888671875,"train/train/tensor_param_model_layers_55_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/mean":-0.00016117095947265625,"train/train/tensor_grad_model_layers_52_mlp_gate_proj_weight/max_abs":0.0016937255859375,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_19_mlp/mean":-0.00017303228378295898,"train/train/tensor_param_model_layers_61_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_58_self_attn_v_proj/frac_near_user_limit":0,"train/train/layer__model_layers_30/param/std":0.044248429256802155,"train/train/tensor_act_model_layers_71_mlp/norm":93.49329162981647,"train/train/tensor_act_model_layers_26_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/mean":1.541338860988617e-07,"train/train/tensor_act_model_layers_44_mlp/max_abs":0.09130859375,"train/train/tensor_act_model_layers_93_self_attn_v_proj/max_abs":1.015625,"train/train/tensor_grad_model_layers_80_mlp_gate_proj_weight/mean":4.284083843231201e-07,"train/train/tensor_act_model_layers_92_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/norm":0.07207029595930536,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_mlp_up_proj/std":0.2265672561181234,"train/train/layer_model_layers_52/grad/mean":-1.7682801788257268e-08,"train/train/tensor_grad_model_layers_44_mlp_gate_proj_weight/std":0.00010290278650369973,"train/train/tensor_grad_model_layers_84_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_37/grad/norm":0.21636581139263625,"train/train/tensor_act_model_layers_45_self_attn/max_abs":0.236328125,"train/train/tensor_grad_model_layers_69_self_attn_o_proj_weight/std":0.0004484635422483906,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/mean":-1.049041748046875e-05,"train/train/tensor_act_model_layers_58_mlp/mean":-6.842613220214844e-05,"train/train/layer_model_layers_59/grad/max_abs":0.005615234375,"train/train/tensor_act_model_layers_0_self_attn/max_abs":0.21484375,"train/train/tensor_act_model_layers_28_self_attn_v_proj/mean":-0.018157958984375,"train/train/tensor_act_model_layers_84_self_attn_q_proj/std":0.22632829866527732,"train/train/tensor_act_model_layers_26_self_attn_k_proj/max_abs":1.0390625,"train/train/tensor_act_model_layers_49_self_attn_k_proj/max_abs":1.1171875,"train/train/tensor_grad_model_layers_54_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_self_attn_v_proj/norm":1330.6883266254856,"train/train/tensor_act_model_layers_78_self_attn_q_proj/std":0.22339212522905627,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_2_mlp_up_proj_weight/mean":-7.963180541992188e-05,"train/train/tensor_act_model_layers_12_self_attn_k_proj/std":0.22315242970383975,"train/train/tensor_act_model_layers_82_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_mlp_up_proj/std":0.22412279163187027,"train/train/tensor_param_model_layers_33_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_81_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_mlp_down_proj/std":0.01568689766045148,"train/train/tensor_grad_model_layers_63_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_91_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/norm":0.0007950670157549293,"train/train/tensor_act_model_layers_90_mlp_gate_proj/max_abs":1.0546875,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/mean":9.754148777574301e-08,"train/train/layer_model_layers_90/grad/std":0.00017596243274637546,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/mean":2.8908252716064453e-06,"train/train/layer_model_layers_82/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/std":0.0006726421783360521,"train/train/tensor_grad_model_layers_71_mlp_gate_proj_weight/std":7.573208180111924e-05,"train/train/tensor_param_model_layers_21_self_attn_o_proj_weight/max_abs":0.07763671875,"train/train/tensor_act_model_layers_32_mlp_down_proj/max_abs":0.0810546875,"train/train/tensor_grad_model_layers_29_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_33_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/max_abs":0.07958984375,"train/train/tensor_grad_model_layers_75_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_67_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_63_post_attention_layernorm_weight/std":3.926323696269653e-05,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/mean":4.3511390686035156e-05,"train/train/layer_model_layers_38/grad/frac_near_dtype_limit":0,"train/train/layer_model_layers_10/act/norm":8932.250505743463,"train/train/tensor_act_model_layers_65_self_attn/mean":0.00017781555652618408,"train/train/layer__model_layers_72/param/mean":0.001620288200199883,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/norm":0.0015380848021730139,"train/train/tensor_act_model_layers_65_input_layernorm/std":1.000005600114272,"train/train/tensor_act_model_layers_90_input_layernorm/norm":5792.592041021804,"train/train/tensor_act_model_layers_6_self_attn_k_proj/std":0.21850978378948718,"train/train/tensor_param_model_layers_26_mlp_up_proj_weight/max_abs":0.08447265625,"train/train/tensor_act_model_layers_50_self_attn/mean":-0.0005951225757598877,"train/train/tensor_act_model_layers_6_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16_mlp/max_abs":0.08056640625,"train/train/layer_model_layers_30/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp/mean":-0.0002014636993408203,"train/train/tensor_grad_model_layers_57_post_attention_layernorm_weight/std":3.5245088654856086e-05,"train/train/tensor_grad_model_layers_29_mlp_gate_proj_weight/max_abs":0.00183868408203125,"train/train/tensor_grad_model_layers_81_input_layernorm_weight/norm":0.001939262427821004,"train/train/tensor_act_model_layers_80_self_attn_o_proj/max_abs":0.232421875,"train/train/tensor_act_model_layers_52_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_59_self_attn_k_proj_weight/max_abs":0.08056640625,"train/train/tensor_param_model_layers_76_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_embed_tokens/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_self_attn/mean":-0.0011587142944335938,"train/train/tensor_act_model_layers_80_mlp_down_proj/max_abs":0.0869140625,"train/train/layer__model_layers_34/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/norm":0.00014645112300998119,"train/train/tensor_act_model_layers_5_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/mean":4.540197551250458e-09,"train/train/tensor_param_model_layers_71_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_92/act/mean":0.0026619434356689453,"train/train/tensor_act_model_layers_45_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_42_mlp/std":0.015429028782923203,"train/train/tensor_act_model_layers_69_mlp_gate_proj/norm":1822.6316269525723,"train/train/layer_model_layers_75/act/norm":9268.037013243928,"train/train/tensor_param_model_layers_30_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/mean":-5.5789947509765625e-05,"train/train/layer_model_layers_91/act/mean":0.0029706954956054688,"train/train/tensor_act_model_layers_44_mlp_down_proj/std":0.016083084552597656,"train/train/tensor_grad_model_layers_85_input_layernorm_weight/norm":0.0015817199580320694,"train/train/layer_model_layers_69/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_26_mlp_down_proj/std":0.01548832222330631,"train/train/tensor_act_model_layers_19_input_layernorm/std":1.000003119925758,"train/train/tensor_act_model_layers_7_post_attention_layernorm/std":0.996103443771765,"train/train/tensor_act_model_layers_63_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_input_layernorm_weight/mean":-1.4938414096832275e-06,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_5_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_26_self_attn_o_proj/mean":0.0011844635009765625,"train/train/layer_model_layers_38/act/std":0.4188791870188775,"train/train/tensor_grad_model_layers_92_self_attn_k_proj_weight/std":4.254814002958003e-07,"train/train/tensor_param_model_layers_52_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_mlp_down_proj_weight/max_abs":0.0908203125,"train/train/tensor_act_model_layers_78_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_4_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_32/max_abs":1.4296875,"train/train/tensor_act_model_layers_74_mlp_down_proj/mean":0.0005035400390625,"train/train/tensor_param_model_layers_58_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_61_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_q_proj_weight/max_abs":6.318092346191406e-06,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp_up_proj/max_abs":1.09375,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/std":0.02001953125,"train/train/layer_model_layers_48/act/max_abs":4.71875,"train/train/tensor_param_model_layers_15_mlp_down_proj_weight/mean":3.3855438232421875e-05,"train/train/tensor_act_model_layers_64_self_attn_o_proj/max_abs":0.236328125,"train/train/tensor_act_model_layers_2_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_mlp_gate_proj_weight/std":8.453648189713516e-05,"train/train/tensor_grad_model_layers_44_self_attn_o_proj_weight/max_abs":0.003936767578125,"train/train/tensor_param_model_layers_19_mlp_down_proj_weight/mean":0.0002498626708984375,"train/train/tensor_act_model_layers_17_mlp_gate_proj/std":0.21753040574601412,"train/train/tensor_param_model_layers_72_mlp_up_proj_weight/mean":-7.772445678710938e-05,"train/train/layer__model_layers_56/param/std":0.044226323191796764,"train/train/tensor_act_model_layers_67_mlp_gate_proj/norm":1863.828224576131,"train/train/tensor_act_model_layers_33_mlp_up_proj/mean":-0.007354736328125,"train/train/tensor_act_model_layers_58_mlp_gate_proj/norm":1819.4742225824064,"train/train/tensor_act_model_layers_92_post_attention_layernorm/mean":0.0169677734375,"train/train/tensor_grad_model_layers_50_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_3_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_65_self_attn_v_proj/std":0.2312050386153194,"train/train/layer_model_layers_46/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_40_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/norm":0.0001279666054305637,"train/train/layer__model_layers_22/param/norm":17.936002768523984,"train/train/tensor_param_model_layers_23_self_attn_k_proj_weight/max_abs":0.09423828125,"train/train/tensor_grad_model_layers_36_self_attn_o_proj_weight/std":0.0005746558802071068,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/norm":0.008731814543154555,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/max_abs":0.07861328125,"train/train/tensor_param_model_layers_65_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp/norm":86.69915456850266,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/mean":-5.145557224750519e-08,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/mean":8.99490260053426e-10,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_8_mlp_up_proj/norm":1875.2965698098415,"train/train/layer_model_layers_82/act/std":0.4287767367045148,"train/train/tensor_act_model_layers_93_post_attention_layernorm/max_abs":4.40625,"train/train/tensor_act_model_layers_35_self_attn/mean":0.0007529258728027344,"train/train/tensor_grad_model_layers_91_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp_down_proj/mean":-8.702278137207031e-06,"train/train/tensor_act_model_layers_25_mlp_up_proj/mean":0.00608062744140625,"train/train/tensor_grad_model_layers_2_mlp_gate_proj_weight/std":0.0005113618571929713,"train/train/tensor_act_model_layers_10_mlp_gate_proj/frac_near_user_limit":0,"train/train/layer_model_layers_45/grad/std":0.00023891410501327332,"train/train/tensor_param_model_layers_54_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/std":0.00013379498472406118,"train/train/tensor_act_model_layers_23_self_attn_k_proj/mean":-0.004146575927734375,"train/train/tensor_act_model_layers_89_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/std":0.045411552251038725,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/norm":0.00015545915207172955,"train/train/tensor_grad_model_layers_80_self_attn_q_proj_weight/max_abs":5.8710575103759766e-06,"train/train/tensor_param_model_layers_60_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_k_proj_weight/mean":6.875779945403337e-10,"train/train/tensor_act_model_layers_49_mlp_up_proj/max_abs":1.125,"train/train/tensor_grad_model_layers_40_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72_mlp_down_proj/max_abs":0.07958984375,"train/train/tensor_act_model_layers_21_mlp_up_proj/norm":1851.3102226686162,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/norm":0.0013743820095349722,"train/train/tensor_act_model_layers_28_mlp_up_proj/std":0.2263194944966397,"train/train/tensor_act_model_layers_32_mlp_down_proj/mean":0.00024127960205078125,"train/train/tensor_act_model_layers_34_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_q_proj_weight/std":0.02001953125,"train/train/layer__model_layers_44/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_34_self_attn_q_proj_weight/max_abs":0.08544921875,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/mean":0.000156402587890625,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/mean":9.182840585708618e-07,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/std":8.404756219352938e-05,"train/train/tensor_grad_model_layers_14_self_attn_v_proj_weight/norm":0.2511973319197452,"train/train/layer_model_layers_89/act/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_mlp_down_proj_weight/max_abs":0.0012664794921875,"train/train/tensor_act_model_layers_52_mlp_down_proj/norm":89.69475723622551,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_71_self_attn_q_proj_weight/std":3.8842272970184735e-07,"train/train/tensor_act_model_layers_91/mean":0.0061187744140625,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_65_mlp_gate_proj/norm":1862.8800412208625,"train/train/tensor_grad_model_layers_50_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_10_mlp_down_proj_weight/max_abs":0.09423828125,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_mlp_down_proj_weight/std":8.117026264768424e-05,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/max_abs":0.0859375,"train/train/tensor_act_model_layers_34_self_attn/max_abs":0.251953125,"train/train/tensor_act_model_layers_9_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/mean":-1.1809170246124268e-06,"train/train/tensor_act_model_layers_49_self_attn_v_proj/mean":0.00849151611328125,"train/train/tensor_param_model_layers_3_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_15_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38/frac_near_user_limit":0,"train/train/layer_model_layers_68/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/norm":0.08641578056104315,"train/train/tensor_param_model_layers_54_self_attn_o_proj_weight/max_abs":0.08056640625,"train/train/tensor_param_model_layers_52_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/norm":0.030931819914529257,"train/train/tensor_grad_model_layers_87_self_attn_v_proj_weight/mean":-2.0619481801986694e-06,"train/train/tensor_param_model_layers_60_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_77/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_o_proj_weight/std":0.001605359522377399,"train/train/layer_model_layers_93/act/max_abs":4.59375,"train/train/tensor_grad_model_layers_27_mlp_up_proj_weight/mean":1.014210283756256e-06,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/max_abs":0.0791015625,"train/train/tensor_grad_model_layers_85_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_4_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp_down_proj/std":0.014969158977573704,"train/train/tensor_act_model_layers_18_self_attn_q_proj/std":0.22070962972254554,"train/train/tensor_act_model_layers_55_input_layernorm/mean":-0.010101318359375,"train/train/tensor_grad_model_layers_32_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn_v_proj/max_abs":1.015625,"train/train/tensor_grad_model_layers_11_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_mlp/mean":0.00012385845184326172,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_5/act/norm":8922.262661608354,"train/train/tensor_act_model_layers_16_self_attn_k_proj/mean":-0.00015854835510253906,"train/train/tensor_act_model_layers_58_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_up_proj_weight/norm":0.23329064341985722,"train/train/tensor_grad_model_layers_5_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_42_self_attn_q_proj/norm":1275.844150467511,"train/train/tensor_param_model_layers_18_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_2_mlp_gate_proj_weight/max_abs":0.09521484375,"train/train/tensor_param_model_layers_68_mlp_down_proj_weight/mean":2.181529998779297e-05,"train/train/tensor_act_model_layers_70_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_input_layernorm/norm":5792.5927734377365,"train/train/tensor_grad_model_layers_67_mlp_up_proj_weight/std":7.213077570237068e-05,"train/train/tensor_param_model_layers_92_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_10_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/max_abs":6.496906280517578e-06,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/max_abs":0.0888671875,"train/train/layer__model_layers_13/param/mean":0.001565202721939444,"train/train/tensor_act_model_layers_22_self_attn_v_proj/std":0.21362703306021247,"train/train/tensor_grad_model_layers_24_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_post_attention_layernorm_weight/max_abs":0.00021457672119140625,"train/train/tensor_act_model_layers_67_self_attn_v_proj/mean":-0.0054779052734375,"train/train/tensor_act_model_layers_89_input_layernorm/max_abs":4.625,"train/train/tensor_param_model_layers_59_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_self_attn_v_proj/std":0.2316981857631783,"train/train/layer_model_layers_87/act/frac_near_user_limit":0,"train/train/layer__model_layers_42/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_mlp_up_proj_weight/max_abs":0.078125,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/max_abs":0.006011962890625,"train/train/tensor_act_model_layers_20_mlp_down_proj/mean":-0.0002014636993408203,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/max_abs":0.004974365234375,"train/train/tensor_act_model_layers_55_mlp/std":0.014969158977573704,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/mean":-2.1827872842550278e-09,"train/train/tensor_param_model_layers_88_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_4_mlp_up_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_mlp_up_proj_weight/mean":-2.62516550719738e-07,"train/train/tensor_act_model_layers_50_self_attn_k_proj/norm":1256.2094691359025,"train/train/tensor_act_model_layers_27_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_input_layernorm_weight/norm":0.0026524630015536134,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/norm":2.59375,"train/train/tensor_grad_model_layers_51_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_83_mlp_down_proj_weight/max_abs":0.08935546875,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_mlp_up_proj/mean":0.0009407997131347656,"train/train/tensor_param_model_layers_14_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_self_attn_o_proj_weight/std":0.001433911837034321,"train/train/tensor_act_model_layers_48_self_attn_o_proj/mean":0.00104522705078125,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/mean":0.000278472900390625,"train/train/tensor_act_model_layers_90_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_38_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_21/grad/std":0.00034661740754351144,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_39/param/mean":0.0015128101461949102,"train/train/tensor_grad_model_layers_86_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_self_attn_v_proj/norm":1313.349875024115,"train/train/tensor_act_model_layers_7_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_mlp_up_proj/max_abs":1.1484375,"train/train/tensor_param_model_layers_35_mlp_up_proj_weight/mean":6.079673767089844e-05,"train/train/tensor_grad_model_layers_86_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/max_abs":0.08349609375,"train/train/tensor_act_model_layers_9_mlp_gate_proj/norm":1856.4125304406614,"train/train/tensor_param_model_layers_14_mlp_up_proj_weight/std":0.02001953125,"train/train/layer_model_layers_73/act/mean":0.0007435934884207589,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/norm":0.0050658720083334935,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/norm":0.03901841537036353,"train/train/tensor_act_model_layers_37_self_attn_v_proj/norm":1315.8267686270383,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/std":0.019775390625,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/std":0.0006457272362200993,"train/train/tensor_param_model_layers_21_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_92_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/norm":0.0021366527684919144,"train/train/tensor_param_model_layers_84_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_68/grad/max_abs":0.005340576171875,"train/train/tensor_act_model_layers_33_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_v_proj/std":0.22949332746770462,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_53_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_mlp_up_proj_weight/max_abs":0.000774383544921875,"train/train/tensor_grad_model_layers_24_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_50_mlp_down_proj/mean":-0.00026535987854003906,"train/train/layer_model_layers_31/act/mean":-0.002484015056065151,"train/train/tensor_grad_model_layers_76_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_39_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_68/grad/std":0.0001968699638489837,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_61_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_51_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_77_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_mlp_up_proj/mean":-0.00838470458984375,"train/train/tensor_param_model_layers_48_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_27_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_mlp_up_proj_weight/max_abs":0.00087738037109375,"train/train/tensor_act_model_layers_19_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_10/mean":-0.008819580078125,"train/train/tensor_act_model_layers_64/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_7_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_v_proj_weight/max_abs":0.0120849609375,"train/train/tensor_grad_model_layers_85_mlp_up_proj_weight/std":6.990751327728742e-05,"train/train/tensor_param_model_layers_38_mlp_gate_proj_weight/max_abs":0.0888671875,"train/train/tensor_act_model_layers_38_mlp/max_abs":0.0888671875,"train/train/tensor_param_model_layers_82_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_81_self_attn_o_proj/max_abs":0.23046875,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/norm":0.026384842224936114,"train/train/tensor_param_model_layers_2_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_45_mlp_down_proj_weight/mean":-0.000213623046875,"train/train/tensor_act_model_layers_74_self_attn_k_proj/mean":-0.017547607421875,"train/train/tensor_act_model_layers_1/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_o_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_75_mlp/mean":-0.00047397613525390625,"train/train/tensor_act_model_layers_52_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/max_abs":0.00157928466796875,"train/train/layer_model_layers_19/act/frac_near_user_limit":0,"train/train/tensor_act_model_layers_64_mlp_down_proj/max_abs":0.08544921875,"train/train/tensor_grad_model_layers_20_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/mean":-1.6944250091910362e-07,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/max_abs":0.006103515625,"train/train/tensor_act_lm_head/norm":7385.928798659407,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_self_attn_v_proj_weight/norm":0.1200149224857588,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_23_mlp_gate_proj/max_abs":1.1796875,"train/train/tensor_grad_model_layers_76_mlp_up_proj_weight/max_abs":0.00136566162109375,"train/train/tensor_param_model_layers_51_self_attn_v_proj_weight/max_abs":0.08203125,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/norm":0.07550618698335994,"train/train/tensor_param_model_layers_9_self_attn_k_proj_weight/max_abs":0.0859375,"train/train/tensor_param_model_layers_41_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_90_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_62_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_7_mlp_down_proj/mean":-0.0002181529998779297,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/norm":0.0029605000142517214,"train/train/tensor_grad_model_layers_0_self_attn_q_proj_weight/mean":-3.2057869248092175e-08,"train/train/tensor_param_model_layers_27_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_87_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_11_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_14_mlp_gate_proj_weight/max_abs":0.0028228759765625,"train/train/tensor_param_model_layers_17_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_92_mlp/mean":-0.000743865966796875,"train/train/tensor_act_model_layers_73_mlp/max_abs":0.08642578125,"train/train/layer_model_layers_67/grad/max_abs":0.003997802734375,"train/train/tensor_grad_model_layers_44_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_91_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_24/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_19_input_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_82_self_attn_k_proj/max_abs":1.0859375,"train/train/tensor_param_model_layers_75_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_46_self_attn_k_proj_weight/max_abs":0.08203125,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/mean":-2.0600855350494385e-06,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_74_self_attn_v_proj/max_abs":1.0703125,"train/train/tensor_grad_model_layers_18_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/mean":-0.000110626220703125,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_mlp_up_proj_weight/max_abs":0.00116729736328125,"train/train/tensor_act_model_layers_50_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn/norm":287.34211252400235,"train/train/tensor_act_model_layers_18_self_attn_o_proj/std":0.04522819097907275,"train/train/tensor_act_model_layers_35_post_attention_layernorm/norm":5792.5794677775675,"train/train/tensor_grad_model_norm_weight/norm":0.02349615704281369,"train/train/layer_model_layers_52/grad/norm":0.17526199119327346,"train/train/tensor_param_model_layers_85_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_65_post_attention_layernorm/mean":0.000640869140625,"train/train/tensor_act_model_layers_46_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_29_self_attn_o_proj_weight/std":0.0005750040581882464,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/std":0.020263671875,"train/train/tensor_act_model_layers_13_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_mlp_down_proj_weight/mean":0.0001220703125,"train/train/tensor_param_model_layers_82_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_52_mlp_down_proj_weight/mean":6.198883056640625e-05,"train/train/tensor_param_model_layers_80_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_input_layernorm/norm":5792.156127932017,"train/train/tensor_grad_model_layers_1_self_attn_k_proj_weight/mean":-3.66562744602561e-08,"train/train/tensor_grad_model_layers_92_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/std":1.1385157623455958e-06,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/mean":-1.691281795501709e-06,"train/train/tensor_grad_model_layers_6_self_attn_q_proj_weight/norm":0.0006448813141608944,"train/train/layer_model_layers_75/grad/mean":-1.3372931996645193e-07,"train/train/tensor_grad_model_layers_76_self_attn_v_proj_weight/norm":0.10899669071738513,"train/train/layer_model_layers_44/act/mean":-0.0016133274350847518,"train/train/tensor_act_model_layers_10_mlp_down_proj/max_abs":0.08154296875,"train/train/layer_model_layers_62/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_q_proj/max_abs":1.15625,"train/train/tensor_grad_model_layers_27_mlp_down_proj_weight/norm":0.04031352433670121,"train/train/tensor_act_model_layers_88_mlp_down_proj/std":0.01550340960988413,"train/train/estimated_remaining_minutes":38.83608542274063,"train/train/tensor_act_model_layers_74_self_attn/norm":288.28196579022557,"train/train/tensor_act_model_layers_19_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_80_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_67_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_16_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/norm":0.0009503305945093622,"train/train/tensor_act_model_layers_68/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_66_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18_input_layernorm/max_abs":4.4375,"train/train/tensor_act_model_layers_22_self_attn_k_proj/mean":0.001911163330078125,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/norm":0.004414592095059277,"train/train/tensor_act_model_layers_19_mlp/norm":89.07891139170341,"train/train/tensor_param_model_layers_86_mlp_gate_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_4_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn_o_proj/max_abs":0.255859375,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/norm":0.00012653896226879407,"train/train/tensor_act_model_layers_26_input_layernorm/std":1.0000010784709599,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/mean":3.4906406654044986e-09,"train/train/tensor_act_model_layers_57_mlp_up_proj/norm":1804.5396405139684,"train/train/tensor_act_model_layers_45/std":0.33057193869863283,"train/train/tensor_act_model_layers_79_mlp_down_proj/mean":0.0002675056457519531,"train/train/tensor_grad_model_layers_54_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/norm":3.59375,"train/train/tensor_grad_model_layers_61_self_attn_v_proj_weight/std":0.00044200556387662154,"train/train/tensor_param_model_layers_18_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_81_post_attention_layernorm/max_abs":4.78125,"train/train/tensor_act_model_layers_29_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/mean":8.079223334789276e-07,"train/train/tensor_grad_model_layers_27_self_attn_q_proj_weight/std":8.151609904690399e-07,"train/train/tensor_param_model_layers_92_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_76_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_grad_model_layers_92_self_attn_q_proj_weight/std":4.3048843449364754e-07,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/max_abs":0.0003910064697265625,"train/train/tensor_act_model_layers_74_mlp_up_proj/max_abs":1.046875,"train/train/layer_model_layers_44/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83/mean":0.003574371337890625,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/mean":1.2278556823730469e-05,"train/train/tensor_act_model_layers_54_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_19_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_32_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_93_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/mean":-1.4659017324447632e-06,"train/train/layer__model_layers_78/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn_k_proj/std":0.225101255094086,"train/train/tensor_param_model_layers_67_mlp_up_proj_weight/mean":-3.743171691894531e-05,"train/train/tensor_act_model_layers_78_self_attn_o_proj/std":0.051271675674611376,"train/train/tensor_grad_model_layers_78_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_68_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_57_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_self_attn_o_proj/std":0.04962322336792662,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/std":6.787673937393348e-05,"train/train/tensor_grad_model_layers_19_input_layernorm_weight/max_abs":0.000946044921875,"train/train/tensor_act_model_layers_30_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn_o_proj/max_abs":0.2080078125,"train/train/tensor_act_model_layers_68_mlp_gate_proj/max_abs":1.1015625,"train/train/tensor_grad_model_layers_65_mlp_up_proj_weight/norm":0.030134373868447552,"train/train/tensor_act_model_layers_92/max_abs":2.375,"train/train/tensor_act_model_layers_48_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/std":9.635273939836712e-05,"train/train/tensor_act_model_layers_40_self_attn_v_proj/norm":1251.9719769119724,"train/train/tensor_act_model_layers_68_self_attn_q_proj/norm":1302.0096378089104,"train/train/layer__model_layers_71/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_87_post_attention_layernorm/max_abs":4.84375,"train/train/tensor_grad_model_layers_92_mlp_up_proj_weight/mean":5.820766091346741e-09,"train/train/tensor_act_model_layers_41/std":0.3154379836008636,"train/train/tensor_grad_model_layers_73_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_76_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_89_mlp_gate_proj_weight/std":6.891329944505032e-05,"train/train/tensor_grad_model_layers_37_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_79/act/mean":0.0026904514857700895,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/mean":5.491077899932861e-06,"train/train/layer_model_layers_55/act/norm":9146.399277209986,"train/train/tensor_param_model_layers_2_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_67_mlp_gate_proj/max_abs":1.2421875,"train/train/tensor_param_model_layers_58_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_o_proj_weight/norm":0.7562075264092832,"train/train/tensor_act_model_layers_45_mlp_gate_proj/mean":-0.00777435302734375,"train/train/tensor_grad_model_layers_13_mlp_down_proj_weight/mean":-4.421919584274292e-06,"train/train/tensor_grad_model_layers_80_mlp_gate_proj_weight/norm":0.025917815701709066,"train/train/tensor_param_model_layers_34_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/norm":0.002734198734869189,"train/train/tensor_act_model_layers_69_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_87_mlp_up_proj_weight/max_abs":0.00156402587890625,"train/train/tensor_act_model_layers_28_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/mean":-1.6748905181884766e-05,"train/train/tensor_act_model_layers_16_mlp/norm":90.95531207695426,"train/train/tensor_act_model_layers_49_mlp/mean":0.00016617774963378906,"train/train/tensor_grad_model_layers_50_self_attn_o_proj_weight/max_abs":0.00482177734375,"train/train/tensor_param_model_layers_23_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_20/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_65_self_attn_k_proj_weight/mean":1.823902130126953e-05,"train/train/tensor_grad_model_layers_18_self_attn_q_proj_weight/max_abs":1.2099742889404297e-05,"train/train/tensor_grad_model_layers_83_mlp_gate_proj_weight/std":7.167448469511913e-05,"train/train/tensor_grad_model_layers_74_mlp_gate_proj_weight/std":7.345080458011913e-05,"train/train/tensor_act_model_layers_36_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_37_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_19_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_24_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_26_input_layernorm_weight/max_abs":0.000736236572265625,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_56_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp_up_proj/mean":0.01092529296875,"train/train/tensor_act_model_layers_11_self_attn_o_proj/mean":-0.0011587142944335938,"train/train/tensor_act_model_layers_62_mlp_up_proj/std":0.22290114687804904,"train/train/tensor_param_model_layers_84_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_59_self_attn_k_proj/max_abs":1.0625,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_9_mlp_up_proj_weight/norm":0.08309752160701908,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_22_input_layernorm/max_abs":4.4375,"train/train/tensor_grad_model_layers_13_self_attn_v_proj_weight/max_abs":0.011474609375,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/norm":0.03436815887347942,"train/train/layer_model_layers_38/grad/mean":2.178123406775284e-07,"train/train/tensor_act_model_layers_33_self_attn_v_proj/max_abs":1.1328125,"train/train/tensor_act_model_layers_41_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_75_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_76_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_88_mlp/mean":-0.0001482069492340088,"train/train/tensor_param_model_layers_61_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_87_self_attn_k_proj/norm":1286.5819239259204,"train/train/tensor_act_model_layers_5_post_attention_layernorm/mean":-0.06512451171875,"train/train/tensor_param_model_layers_35_self_attn_v_proj_weight/max_abs":0.07421875,"train/train/layer__model_layers_61/param/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_mlp_up_proj/std":0.2275406049248394,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/max_abs":0.003631591796875,"train/train/tensor_act_model_layers_67_self_attn_k_proj/max_abs":1.1015625,"train/train/tensor_act_model_layers_81_input_layernorm/std":1.0000005541367785,"train/train/tensor_param_model_layers_93_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/std":5.993871752928878e-07,"train/train/tensor_act_model_layers_52_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_q_proj_weight/std":8.66394839953461e-07,"train/train/tensor_act_model_layers_86_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_mlp_gate_proj_weight/mean":-3.043562173843384e-06,"train/train/tensor_act_model_layers_7_mlp/norm":90.9582977715985,"train/train/tensor_act_model_layers_62_self_attn_o_proj/std":0.05066094740115944,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/max_abs":0.0001583099365234375,"train/train/tensor_param_model_layers_82_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_v_proj/mean":-0.0083465576171875,"train/train/tensor_param_model_layers_36_self_attn_v_proj_weight/max_abs":0.0859375,"train/train/tensor_param_model_layers_90_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23/mean":-0.011932373046875,"train/train/tensor_act_model_layers_69_mlp_up_proj/std":0.22681080174812224,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/mean":-2.250075340270996e-06,"train/train/tensor_grad_model_layers_31_self_attn_v_proj_weight/max_abs":0.00750732421875,"train/train/tensor_param_model_layers_58_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_12_mlp_up_proj_weight/max_abs":0.002655029296875,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/norm":2.59375,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_62_self_attn_q_proj/max_abs":1.046875,"train/train/tensor_act_model_layers_77_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_o_proj/max_abs":0.23046875,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_23_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_77_self_attn_v_proj_weight/std":0.00043771812100098644,"train/train/tensor_param_model_layers_70_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_90_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_o_proj_weight/max_abs":0.004547119140625,"train/train/tensor_act_model_layers_51_self_attn_v_proj/max_abs":1.03125,"train/train/tensor_grad_model_layers_9_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_k_proj/mean":0.00582122802734375,"train/train/tensor_act_model_layers_81_post_attention_layernorm/norm":5792.601074223483,"train/train/tensor_param_model_layers_66_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_27_mlp_down_proj/max_abs":0.08154296875,"train/train/tensor_param_model_layers_39_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_self_attn/std":0.04760935711442432,"train/train/tensor_grad_model_layers_53_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/norm":0.04117878749083183,"train/train/tensor_act_model_layers_45_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_mlp_down_proj_weight/max_abs":0.0035858154296875,"train/train/tensor_act_model_layers_78/std":0.44433977397851276,"train/train/tensor_grad_model_layers_63_mlp_gate_proj_weight/max_abs":0.00146484375,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/mean":-2.8848648071289062e-05,"train/train/tensor_param_model_layers_3_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/mean":-5.245208740234375e-05,"train/train/tensor_grad_model_layers_47_self_attn_v_proj_weight/std":0.0004501276902965552,"train/train/tensor_param_model_layers_43_mlp_up_proj_weight/max_abs":0.08349609375,"train/train/tensor_param_model_layers_89_mlp_gate_proj_weight/std":0.02001953125,"train/train/layer_model_layers_3/grad/std":0.0011110694465120098,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_86/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_mlp_down_proj_weight/max_abs":0.083984375,"train/train/tensor_param_model_layers_28_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10/std":0.14917111123168095,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_74/param/mean":0.0015883423422874414,"train/train/tensor_grad_model_layers_67_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_72/max_abs":2.109375,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/std":4.755296123610255e-07,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/max_abs":0.09228515625,"train/train/tensor_act_model_layers_93_mlp_gate_proj/mean":0.0017108917236328125,"train/train/tensor_act_model_layers_60_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/max_abs":4.589557647705078e-06,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp_gate_proj/mean":-0.0012056827545166016,"train/train/tensor_param_model_layers_18_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_self_attn_o_proj_weight/max_abs":0.08203125,"train/train/tensor_act_model_layers_71_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_59_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_mlp_up_proj/mean":0.0102996826171875,"train/train/tensor_act_model_layers_50_post_attention_layernorm/std":1.0000063765925429,"train/train/layer__model_layers_18/param/max_abs":1,"train/train/tensor_act_model_layers_90_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_mlp_up_proj_weight/mean":-4.016328603029251e-09,"train/train/tensor_param_model_layers_68_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_63_self_attn_k_proj/mean":-0.003734588623046875,"train/train/tensor_act_model_layers_89_mlp_down_proj/mean":0.0006132125854492188,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_mlp/std":0.015458980657959814,"train/train/layer_model_layers_30/grad/std":0.00029050180174505113,"train/train/tensor_act_model_layers_51_self_attn_q_proj/std":0.2326737384575181,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/mean":5.936622619628906e-05,"train/train/layer_model_layers_52/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_mlp_down_proj/max_abs":0.07958984375,"train/train/tensor_grad_model_layers_22_self_attn_o_proj_weight/max_abs":0.00555419921875,"train/train/tensor_grad_model_layers_68_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_self_attn_q_proj_weight/mean":-1.1444091796875e-05,"train/train/tensor_param_model_layers_36_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_self_attn_o_proj_weight/mean":2.4557113647460938e-05,"train/train/tensor_act_model_layers_85_mlp_down_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_70/act/norm":9237.778585991808,"train/train/tensor_act_model_layers_55_input_layernorm/norm":5792.58337402765,"train/train/tensor_act_model_layers_38_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_q_proj/mean":0.00555419921875,"train/train/layer_model_layers_54/act/std":0.4218908585486412,"train/train/tensor_grad_model_layers_15_mlp_up_proj_weight/norm":0.06383250283364186,"train/train/tensor_act_model_layers_82_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_mlp_gate_proj/std":0.23120227373220878,"train/train/tensor_param_model_layers_49_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_40_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_7_self_attn_k_proj_weight/max_abs":0.083984375,"train/train/layer_model_layers_15/grad/std":0.000440525588202105,"train/train/tensor_act_model_layers_71_self_attn_o_proj/max_abs":0.2890625,"train/train/tensor_act_model_layers_33_self_attn/mean":0.0007238388061523438,"train/train/tensor_act_model_layers_24_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_28/max_abs":1.28125,"train/train/tensor_act_model_layers_54_self_attn_k_proj/std":0.2189994155393097,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/max_abs":0.00135040283203125,"train/train/tensor_param_model_layers_44_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_grad_model_layers_47_mlp_up_proj_weight/norm":0.03480909394564197,"train/train/tensor_param_model_layers_66_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_46_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_64/act/max_abs":4.71875,"train/train/tensor_param_model_layers_27_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_45_mlp_down_proj/max_abs":0.0849609375,"train/train/tensor_act_model_layers_51/norm":2045.9104471432124,"train/train/tensor_act_model_layers_49_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_49_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_self_attn_q_proj/mean":0.011810302734375,"train/train/tensor_act_model_layers_4_input_layernorm/mean":-0.03997802734375,"train/train/tensor_grad_model_layers_38_mlp_up_proj_weight/norm":0.034800531244059685,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/max_abs":0.08544921875,"train/train/layer_model_layers_15/grad/mean":1.6086258273135853e-06,"train/train/tensor_param_model_layers_39_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_post_attention_layernorm/std":1.0000070581196383,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_53_self_attn/norm":268.64735670760024,"train/train/tensor_act_model_layers_13_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/norm":0.48889999858679084,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/norm":0.00010916108619471522,"train/train/tensor_grad_model_layers_55_post_attention_layernorm_weight/std":3.800386927888027e-05,"train/train/tensor_grad_model_layers_53_self_attn_k_proj_weight/mean":1.997705112444237e-09,"train/train/tensor_act_model_layers_43_mlp/mean":0.0009212493896484375,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_72_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_58_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_post_attention_layernorm_weight/max_abs":0.0003757476806640625,"train/train/tensor_act_model_layers_78_mlp_up_proj/max_abs":1.015625,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/std":9.821111586539027e-05,"train/train/layer_model_layers_44/grad/max_abs":0.005462646484375,"train/train/tensor_param_model_layers_47_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/max_abs":0.0888671875,"train/train/tensor_act_model_layers_84_input_layernorm/mean":0.00742340087890625,"train/train/tensor_act_model/norm":5792.595336918584,"train/train/tensor_param_model_layers_42_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_45_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_70_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_69_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_71_self_attn_v_proj/norm":1346.5370320488712,"train/train/tensor_param_model_layers_78_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_62_mlp/norm":89.86351787799238,"train/train/tensor_act_model_layers_82/norm":2639.3818863998695,"train/train/tensor_act_model_layers_17_self_attn/frac_near_user_limit":0,"train/train/layer_model_layers_83/grad/max_abs":0.00421142578125,"train/train/layer__model_layers_20/param/norm":17.926533287368336,"train/train/tensor_grad_model_layers_53_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_38_mlp_up_proj_weight/max_abs":0.0849609375,"train/train/tensor_grad_model_layers_63_self_attn_k_proj_weight/max_abs":5.27501106262207e-06,"train/train/tensor_param_model_layers_54_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_35_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_37_mlp/norm":92.1369030607865,"train/train/tensor_grad_model_layers_27_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_80_self_attn/frac_near_user_limit":0,"train/train/layer__model_layers_83/param/std":0.04425536969917314,"train/train/tensor_grad_model_layers_52_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_q_proj_weight/std":4.964191247141762e-07,"train/train/tensor_param_model_layers_77_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_12_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_self_attn_q_proj/max_abs":1.2109375,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/mean":-2.6404857635498047e-05,"train/train/tensor_act_model_layers_74_self_attn/mean":-6.699562072753906e-05,"train/train/tensor_act_model_layers_82_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_o_proj_weight/norm":0.27866831656093816,"train/train/tensor_act_model_layers_74_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_49_mlp_up_proj_weight/mean":-1.1801719665527344e-05,"train/train/layer_model_layers_91/grad/mean":1.9650488609503287e-07,"train/train/tensor_param_model_layers_39_mlp_gate_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_35_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_44_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_30_mlp_gate_proj_weight/max_abs":0.083984375,"train/train/layer__model_layers_72/param/max_abs":1,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_26_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_1_input_layernorm/mean":-0.027587890625,"train/train/tensor_grad_model_layers_54_input_layernorm_weight/std":8.879770026368398e-05,"train/train/tensor_act_model_layers_74_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_39_post_attention_layernorm/std":1.0000037974605875,"train/train/tensor_param_model_layers_31_mlp_up_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_61_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/max_abs":0.078125,"train/train/tensor_act_model_layers_90_input_layernorm/mean":0.010528564453125,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_40_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_59/act/max_abs":4.65625,"train/train/tensor_act_model_layers_26_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_self_attn/std":0.052004333364874956,"train/train/tensor_act_model_layers_1/max_abs":0.333984375,"train/train/tensor_act_model_layers_20_self_attn_q_proj/std":0.2392624621036821,"train/train/tensor_grad_model_layers_64_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_34_post_attention_layernorm/std":1.0000028870958135,"train/train/tensor_grad_model_layers_85_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_74_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_74_mlp_down_proj_weight/mean":-3.0040740966796875e-05,"train/train/tensor_act_model_layers_31_post_attention_layernorm/mean":-0.0333251953125,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/mean":-0.00014019012451171875,"train/train/global/param/mean":0.0015211859032954781,"train/train/tensor_act_model_layers_33_mlp/max_abs":0.0849609375,"train/train/layer_model_layers_72/act/std":0.4262275150663574,"train/train/tensor_grad_model_layers_69_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_mlp/max_abs":0.0859375,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_92/norm":2802.4379275438705,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/max_abs":0.00135040283203125,"train/train/tensor_grad_model_layers_88_mlp_up_proj_weight/norm":0.023569846012403885,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_32_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_84_mlp_down_proj/max_abs":0.0830078125,"train/train/tensor_param_model_layers_26_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_27_self_attn_v_proj_weight/std":0.0006583848718483477,"train/train/tensor_grad_model_layers_20_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_62_post_attention_layernorm/norm":5792.590087891947,"train/train/tensor_param_model_layers_72_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_73_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_53_post_attention_layernorm_weight/std":3.916004749140382e-05,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_61_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/norm":0.00018626824375234913,"train/train/tensor_act_model_layers_7_mlp/mean":-0.0002181529998779297,"train/train/layer_model_layers_34/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_self_attn_v_proj_weight/max_abs":0.0771484375,"train/train/tensor_act_model_layers_36_self_attn_v_proj/norm":1328.5282573922284,"train/train/tensor_param_model_layers_46_mlp_gate_proj_weight/max_abs":0.07861328125,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/mean":-1.1241354513913393e-09,"train/train/tensor_param_model_layers_11_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/norm":0.00011094690162700593,"train/train/tensor_param_model_layers_31_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/mean":-2.0474999473663047e-09,"train/train/tensor_grad_model_layers_19_mlp_gate_proj_weight/max_abs":0.00213623046875,"train/train/tensor_param_model_layers_81_mlp_down_proj_weight/max_abs":0.083984375,"train/train/layer__model_layers_9/param/max_abs":1,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_79_self_attn_q_proj_weight/max_abs":4.172325134277344e-06,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_v_proj_weight/std":0.0011576977916092175,"train/train/tensor_act_model_layers_50_self_attn_q_proj/norm":1311.8728884219342,"train/train/tensor_grad_model_layers_60_self_attn_o_proj_weight/max_abs":0.003997802734375,"train/train/tensor_act_model_layers_61_mlp_up_proj/norm":1838.3550360434313,"train/train/tensor_param_model_layers_88_mlp_up_proj_weight/mean":-5.316734313964844e-05,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/max_abs":0.07568359375,"train/train/tensor_act_model_layers_78_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_32_mlp_up_proj_weight/mean":-4.4345855712890625e-05,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/mean":-4.0531158447265625e-05,"train/train/tensor_act_model_layers_83_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_13/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_83_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/mean":9.601935744285583e-07,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/norm":0.0026564539452404038,"train/train/layer_model_layers_60/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/mean":1.001353666651994e-09,"train/train/tensor_grad_model_layers_45_mlp_up_proj_weight/max_abs":0.00148773193359375,"train/train/tensor_grad_model_layers_65_self_attn_q_proj_weight/norm":0.00011794203474455115,"train/train/tensor_act_model_layers_69_mlp_up_proj/norm":1859.6205793365145,"train/train/tensor_grad_model_layers_69_self_attn_v_proj_weight/norm":0.11063072879419142,"train/train/tensor_param_model_layers_38_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/norm":2.5625,"train/train/layer__model_layers_4/param/max_abs":1,"train/train/tensor_param_model_layers_45_mlp_gate_proj_weight/mean":-9.1552734375e-05,"train/train/tensor_act_model_layers_63_self_attn/mean":-0.0013637542724609375,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_2_self_attn_v_proj/max_abs":1.3046875,"train/train/layer_model_layers_57/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_input_layernorm_weight/std":6.758063289444058e-05,"train/train/tensor_param_model_layers_60_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_52_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/mean":-4.105459083802998e-09,"train/train/tensor_act_model_layers_42_self_attn_k_proj/norm":1339.0173674938953,"train/train/tensor_act_model_layers_12/mean":-0.00811767578125,"train/train/tensor_grad_model_layers_39_self_attn_k_proj_weight/norm":0.00014984982295425882,"train/train/tensor_grad_model_layers_40_input_layernorm_weight/max_abs":0.000591278076171875,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp/std":0.015717023452114582,"train/train/tensor_act_model_layers_2_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_40_self_attn/std":0.04785333281643411,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/mean":-1.0099029168486595e-07,"train/train/tensor_param_model_layers_10_self_attn_v_proj_weight/mean":8.96453857421875e-05,"train/train/layer_model_layers_87/act/frac_near_dtype_limit":0,"train/train/layer__model_layers_5/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_post_attention_layernorm_weight/norm":11.3125,"train/train/layer_model_layers_7/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_48_mlp/norm":87.62625324793763,"train/train/tensor_param_model_layers_52_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_31_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_35_self_attn_q_proj_weight/mean":8.535385131835938e-05,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/std":0.019775390625,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_self_attn_o_proj_weight/max_abs":0.08154296875,"train/train/tensor_act_model_layers_21_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_68_mlp/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_13_mlp_up_proj_weight/mean":1.1548399925231934e-06,"train/train/tensor_act_model_layers_79_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_66/act/max_abs":4.53125,"train/train/tensor_act_model_layers_9/max_abs":0.73828125,"train/train/tensor_grad_model_layers_73_mlp_gate_proj_weight/mean":3.186287358403206e-07,"train/train/tensor_grad_model_layers_9_self_attn_o_proj_weight/std":0.0012268288959971154,"train/train/tensor_act_model_layers_65_post_attention_layernorm/max_abs":4.53125,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/norm":0.00013251012069311383,"train/train/tensor_act_model_layers_4_input_layernorm/max_abs":5,"train/train/tensor_param_model_layers_55_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_18_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/std":0.00010871951745860212,"train/train/tensor_act_model_layers_40_self_attn/max_abs":0.2265625,"train/train/layer__model_layers_3/param/mean":0.0016035773266868175,"train/train/tensor_grad_model_layers_26_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/mean":5.64978108741343e-09,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/mean":2.857297658920288e-06,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/mean":-6.891787052154541e-07,"train/train/tensor_param_model_layers_72_input_layernorm_weight/norm":11.3125,"train/train/tensor_param_model_layers_84_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_53_self_attn_o_proj_weight/norm":0.10544981609923532,"train/train/tensor_param_model_layers_78_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/max_abs":0.000400543212890625,"train/train/tensor_act_model_layers_55_post_attention_layernorm/mean":-0.005546055734157562,"train/train/tensor_grad_model_layers_4_self_attn_q_proj_weight/std":3.7122739080674613e-06,"train/train/tensor_act_model_layers_83_mlp_gate_proj/mean":-0.0019731521606445312,"train/train/layer_model_layers_0/act/std":0.4105403647997667,"train/train/tensor_param_model_layers_62_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_28_self_attn_v_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_63_mlp_up_proj_weight/mean":4.752073436975479e-07,"train/train/layer__model_layers_70/param/mean":0.0016607799321738494,"train/train/tensor_act_model_layers_14_self_attn_v_proj/norm":1338.13869440254,"train/train/tensor_param_model_layers_3_mlp_down_proj_weight/max_abs":0.08251953125,"train/train/tensor_act_model_layers_81_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_85_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/mean":-5.034962669014931e-09,"train/train/tensor_grad_model_layers_25_input_layernorm_weight/mean":3.3229589462280273e-06,"train/train/tensor_grad_model_layers_69_mlp_gate_proj_weight/mean":2.8777867555618286e-07,"train/train/layer__model_layers_37/param/norm":17.943644145891184,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_30_self_attn_q_proj/norm":1296.9547163429613,"train/train/tensor_act_model_layers_13_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_43_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_78_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_15_self_attn_k_proj_weight/norm":0.00029309004517801874,"train/train/layer_model_layers_37/grad/max_abs":0.006103515625,"train/train/tensor_act_model_layers_43_self_attn_q_proj/std":0.22193151771224354,"train/train/tensor_param_model_layers_41_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_4_mlp_gate_proj_weight/std":0.00032587683335784926,"train/train/tensor_grad_model_layers_54_mlp_down_proj_weight/std":8.54025298050208e-05,"train/train/tensor_grad_model_layers_15_self_attn_q_proj_weight/std":1.25390363882325e-06,"train/train/tensor_act_model_layers_54_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_q_proj/norm":1310.4102053732893,"train/train/tensor_act_model_layers_18_self_attn_k_proj/frac_near_user_limit":0,"train/train/layer_model_layers_30/grad/mean":1.3998175125609786e-07,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/max_abs":0.0791015625,"train/train/tensor_param_model_layers_35_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_37_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_0_input_layernorm/norm":5785.478271494312,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/norm":0.0012183404946706534,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/norm":0.045462957059058895,"train/train/tensor_param_model_layers_92_mlp_gate_proj_weight/norm":3.609375,"train/train/tensor_act_model_layers_61_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_input_layernorm_weight/norm":0.004096302849461693,"train/train/tensor_grad_model_layers_81_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_71_self_attn_k_proj_weight/max_abs":0.07275390625,"train/train/layer__model_layers_49/param/norm":17.930999747730883,"train/train/layer_model_layers_33/act/mean":-0.005899361201695034,"train/train/tensor_act_model_layers_62_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_3_self_attn_v_proj/mean":0.00023984909057617188,"train/train/tensor_grad_model_layers_5_mlp_gate_proj_weight/mean":3.3490359783172607e-06,"train/train/tensor_param_model_layers_73_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/mean":-8.487701416015625e-05,"train/train/tensor_param_model_layers_50_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_29_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/max_abs":0.00016117095947265625,"train/train/tensor_act_model_layers_86_mlp_gate_proj/norm":1884.6261529117596,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/norm":0.0008072479890600509,"train/train/tensor_grad_model_layers_4_self_attn_o_proj_weight/mean":-7.569789886474609e-06,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/norm":0.10600230572549973,"train/train/layer_model_layers_90/grad/frac_near_user_limit":0,"train/train/layer_model_layers_83/grad/std":0.00016883912273662186,"train/train/layer__model_layers_87/param/norm":17.931387787027248,"train/train/tensor_grad_model_layers_71_self_attn_v_proj_weight/max_abs":0.0040283203125,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_64_mlp_gate_proj/max_abs":1.1796875,"train/train/tensor_act_model_layers_4_mlp/norm":92.58727412352023,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/mean":2.1676532924175262e-07,"train/train/tensor_param_model_layers_83_mlp_gate_proj_weight/max_abs":0.08203125,"train/train/tensor_param_model_layers_13_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_31_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_91/act/std":0.4303072060912596,"train/train/tensor_act_model_layers_21_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_93_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_26_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/layer_model_layers_18/grad/mean":2.1381353793808155e-06,"train/train/tensor_act_model_layers_38_self_attn_v_proj/max_abs":1.140625,"train/train/tensor_act_model_layers_81_self_attn_k_proj/max_abs":1.015625,"train/train/tensor_act_model_layers_56_input_layernorm/max_abs":4.90625,"train/train/tensor_grad_model_layers_44_self_attn_q_proj_weight/max_abs":4.500150680541992e-06,"train/train/tensor_act_model_layers_58_self_attn_q_proj/std":0.22852903463274316,"train/train/tensor_act_model_layers_59_self_attn/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_param_model_layers_19_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_mlp_gate_proj_weight/max_abs":0.00128173828125,"train/train/tensor_param_model_layers_66_mlp_gate_proj_weight/mean":-0.00016117095947265625,"train/train/tensor_act_model_layers_30_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_9_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/mean":-1.914333552122116e-06,"train/train/tensor_act_model_layers_45_input_layernorm/norm":5792.5845947294065,"train/train/tensor_act_model_layers_0_self_attn/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/norm":0.1590373332980919,"train/train/tensor_act_model_layers_4_mlp_down_proj/mean":-0.0007181167602539062,"train/train/tensor_param_model_layers_46_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_post_attention_layernorm_weight/max_abs":0.001495361328125,"train/train/tensor_param_model_layers_3_self_attn_v_proj_weight/max_abs":0.0888671875,"train/train/tensor_act_model_layers_93_self_attn_v_proj/std":0.2297386925359787,"train/train/tensor_grad_model_layers_82_self_attn_v_proj_weight/std":0.00035174331309737215,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/mean":5.3085386753082275e-08,"train/train/tensor_act_model_layers_14_mlp/norm":91.64584453358795,"train/train/tensor_act_model_layers_18_mlp/mean":0.00020074844360351562,"train/train/tensor_grad_model_layers_39_self_attn_v_proj_weight/norm":0.12539134753247827,"train/train/tensor_grad_model_layers_82_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/max_abs":0.00018024444580078125,"train/train/tensor_param_model_layers_92_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_56_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_39_self_attn_o_proj_weight/max_abs":0.0047607421875,"train/train/tensor_act_model_layers_8_mlp/norm":94.78675581007263,"train/train/tensor_grad_model_layers_81_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45/mean":-0.0022268295288085938,"train/train/tensor_act_model_layers_12_mlp_gate_proj/max_abs":1.1640625,"train/train/tensor_grad_model_layers_44_mlp_up_proj_weight/std":0.00010291714116660304,"train/train/tensor_grad_model_layers_41_post_attention_layernorm_weight/norm":0.0009773109965035187,"train/train/tensor_param_model_layers_5_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_post_attention_layernorm_weight/mean":6.809830665588379e-06,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/mean":3.736204234883189e-09,"train/train/tensor_act_model_layers_63_self_attn_o_proj/mean":-0.0013637542724609375,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/mean":-9.067356586456299e-06,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_12_self_attn_k_proj_weight/max_abs":0.07763671875,"train/train/tensor_grad_model_layers_51_mlp_gate_proj_weight/std":8.047305001595725e-05,"train/train/tensor_grad_model_layers_80_self_attn_k_proj_weight/norm":0.0001195097950370778,"train/train/tensor_grad_model_layers_84_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_81_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_13_self_attn_o_proj/std":0.048037360582338394,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/mean":6.645917892456055e-05,"train/train/tensor_grad_model_layers_84_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/mean":-4.1443854570388794e-08,"train/train/tensor_act_model_layers_58_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_77_self_attn_v_proj_weight/std":0.02001953125,"train/train/layer__model_layers_63/param/norm":17.925607174601506,"train/train/tensor_act_model_layers_65_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_self_attn_v_proj/max_abs":1.1015625,"train/train/tensor_act_model_layers_30_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_59_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_mlp_gate_proj_weight/std":7.931177055140057e-05,"train/train/tensor_act_model_layers_35_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_18/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_41_mlp_gate_proj/mean":-0.01214599609375,"train/train/layer_model_layers_36/act/mean":-0.003244715077536447,"train/train/tensor_param_model_layers_78_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_72_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_66_mlp_up_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_59_input_layernorm/mean":-0.0052490234375,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/max_abs":0.0003814697265625,"train/train/layer__model_layers_76/param/std":0.044241502395468056,"train/train/tensor_param_model_layers_52_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_53_self_attn_k_proj/norm":1358.5193846646425,"train/train/layer_model_layers_44/act/max_abs":4.90625,"train/train/tensor_grad_model_layers_42_self_attn_q_proj_weight/std":5.926903463493516e-07,"train/train/tensor_param_model_layers_92_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_62_self_attn_o_proj/mean":0.00325775146484375,"train/train/tensor_act_model_layers_6_self_attn_q_proj/mean":0.021514892578125,"train/train/tensor_act_model_layers_37_mlp_down_proj/max_abs":0.0986328125,"train/train/tensor_act_model_layers_44_mlp/mean":0.00018405914306640625,"train/train/tensor_act_model_layers_67_self_attn_o_proj/max_abs":0.22265625,"train/train/tensor_grad_model_layers_80_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/mean":-0.0001850128173828125,"train/train/tensor_grad_model_layers_21_mlp_down_proj_weight/std":0.00013351702233209883,"train/train/tensor_grad_model_layers_23_mlp_up_proj_weight/norm":0.0496973114403251,"train/train/tensor_param_model_layers_26_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_10_self_attn_o_proj_weight/std":0.019775390625,"train/train/tensor_act_model_layers_16_post_attention_layernorm/mean":-0.05865478515625,"train/train/tensor_act_model_layers_22_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_17_mlp_up_proj/frac_near_user_limit":0,"train/train/layer_model_layers_42/grad/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_92_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_6/grad/std":0.0007523202589619292,"train/train/tensor_act_model_layers_13_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_59_self_attn_o_proj/max_abs":0.2255859375,"train/train/tensor_grad_model_layers_80_self_attn_o_proj_weight/max_abs":0.00390625,"train/train/layer__model_layers_33/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_29/norm":1540.3236234927315,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_41/param/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_78_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_50/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_48_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/mean":3.618188202381134e-07,"train/train/tensor_act_model_layers_56_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_self_attn_q_proj_weight/std":3.1944775937758623e-06,"train/train/tensor_act_model_layers_91_mlp/std":0.015687010166326013,"train/train/tensor_act_model_layers_75_self_attn_o_proj/norm":281.61714924777203,"train/train/tensor_act_model_layers_44_self_attn_k_proj/std":0.22168644924186354,"train/train/tensor_act_model_layers_70_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_6/act/norm":8897.637014261541,"train/train/tensor_param_model_layers_56_mlp_down_proj_weight/mean":-8.630752563476562e-05,"train/train/tensor_param_model_layers_70_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_mlp_gate_proj/mean":-0.003925323486328125,"train/train/tensor_grad_model_layers_63_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_69_mlp_gate_proj/max_abs":1.15625,"train/train/tensor_act_model_layers_71_post_attention_layernorm/max_abs":4.71875,"train/train/tensor_act_model_layers_67_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_mlp_up_proj_weight/mean":-3.3248215913772583e-07,"train/train/tensor_grad_model_layers_57_input_layernorm_weight/max_abs":0.00058746337890625,"train/train/tensor_param_model_layers_55_post_attention_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_64_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_41/param/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_48_self_attn_k_proj_weight/max_abs":1.1146068572998047e-05,"train/train/tensor_act_model_layers_87_self_attn_k_proj/max_abs":1.1875,"train/train/tensor_act_model_layers_31_input_layernorm/std":1.0000000894069632,"train/train/layer_model_layers_86/grad/norm":0.14652362226235338,"train/train/tensor_act_model_layers_79_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_73_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_20_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_83_post_attention_layernorm_weight/std":3.0606056135940654e-05,"train/train/tensor_grad_model_layers_16_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_16/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_47_post_attention_layernorm/std":1.0000056258473708,"train/train/tensor_param_model_layers_17_self_attn_k_proj_weight/max_abs":0.0830078125,"train/train/tensor_act_model_layers_85_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_24_self_attn_k_proj/max_abs":1.109375,"train/train/tensor_act_model_layers_21_input_layernorm/mean":-0.058837890625,"train/train/tensor_act_model_layers_34_mlp_down_proj/norm":90.84844103582273,"train/train/tensor_param_model_layers_53_mlp_gate_proj_weight/max_abs":0.080078125,"train/train/layer_model_layers_67/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_post_attention_layernorm_weight/std":4.469327782537464e-05,"train/train/tensor_act_model_layers_21_self_attn/max_abs":0.224609375,"train/train/tensor_param_model_layers_68_mlp_gate_proj_weight/norm":3.625,"train/train/layer_model_layers_52/act/std":0.42145447074488057,"train/train/tensor_param_model_layers_54_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_5_self_attn_k_proj/norm":1302.6024981950363,"train/train/tensor_grad_model_layers_73_self_attn_q_proj_weight/max_abs":3.814697265625e-06,"train/train/tensor_act_model_layers_21_mlp_up_proj/std":0.22583280545598242,"train/train/tensor_param_model_layers_86_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_29_mlp_gate_proj/max_abs":1.09375,"train/train/tensor_param_model_layers_6_mlp_up_proj_weight/norm":3.609375,"train/train/layer_model_layers_50/grad/mean":-8.30890068511207e-08,"train/train/tensor_act_model_layers_61_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_65_self_attn_q_proj/max_abs":1.046875,"train/train/tensor_grad_model_layers_62_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_85/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_34_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_79_self_attn_v_proj/mean":0.004436492919921875,"train/train/tensor_grad_model_layers_17_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_93_self_attn_k_proj/norm":1287.5175924405582,"train/train/tensor_act_model_layers_50_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_88_mlp_gate_proj/norm":1851.1056338308795,"train/train/tensor_act_model_layers_19_self_attn_o_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_47_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_17_post_attention_layernorm_weight/mean":9.257346391677856e-07,"train/train/tensor_act_model_layers_71_self_attn_k_proj/max_abs":1.1171875,"train/train/tensor_act_model_layers_91/max_abs":2.375,"train/train/tensor_param_model_layers_13_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76_self_attn_v_proj/max_abs":1.125,"train/train/tensor_act_model_layers_47_mlp_up_proj/norm":1839.4369946248862,"train/train/tensor_act_model_layers_35_post_attention_layernorm/max_abs":5.125,"train/train/tensor_param_model_layers_26_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_27_self_attn/std":0.047734447347643524,"train/train/tensor_act_model_layers_81_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/layer__model_layers_8/param/max_abs":1,"train/train/tensor_act_model_layers_27_self_attn_o_proj/std":0.047734447347643524,"train/train/tensor_act_model_layers_20_self_attn/mean":-0.000804901123046875,"train/train/tensor_act_model_layers_18_post_attention_layernorm/max_abs":4.5625,"train/train/tensor_act_model_layers_1/norm":210.41273893667594,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_32_input_layernorm/std":1.000000737607207,"train/train/tensor_act_model_layers_74_mlp_gate_proj/mean":0.00836944580078125,"train/train/tensor_param_model_layers_39_self_attn_v_proj_weight/max_abs":0.08056640625,"train/train/tensor_act_model_layers_87_self_attn_v_proj/std":0.22413081218050945,"train/train/tensor_grad_model_layers_35_mlp_gate_proj_weight/norm":0.04574478076910218,"train/train/tensor_act_model_layers_10_self_attn_q_proj/mean":-0.006832122802734375,"train/train/tensor_param_model_layers_43_mlp_down_proj_weight/max_abs":0.08447265625,"train/train/tensor_act_model_layers_21_self_attn/std":0.046449112958302216,"train/train/layer_model_layers_58/act/mean":-0.0007553441183907646,"train/train/tensor_param_model_layers_80_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/mean":0.00024318695068359375,"train/train/tensor_param_model_layers_51_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_84/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_0_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_7_input_layernorm/max_abs":5.21875,"train/train/tensor_grad_model_layers_31_mlp_gate_proj_weight/std":0.00012021053756948983,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/norm":0.00011686061786433308,"train/train/tensor_grad_model_layers_41_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_52_self_attn_k_proj_weight/mean":-6.83940015733242e-10,"train/train/tensor_grad_model_layers_19_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_90/act/max_abs":4.5625,"train/train/tensor_act_model_layers_33_mlp_gate_proj/std":0.22461211007364268,"train/train/tensor_act_model_layers_30_post_attention_layernorm/std":1.00000031851227,"train/train/tensor_act_model_layers_5_self_attn/std":0.045411552251038725,"train/train/tensor_param_model_layers_57_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_76_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_89/grad/mean":1.6110248274537452e-07,"train/train/tensor_param_model_layers_9_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_70_self_attn_q_proj/norm":1321.8397541734994,"train/train/tensor_grad_model_layers_88_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_22_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_87_mlp_gate_proj_weight/norm":0.0237227407674102,"train/train/tensor_param_model_layers_74_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_38_input_layernorm_weight/max_abs":0.000644683837890625,"train/train/tensor_grad_model_layers_46_self_attn_v_proj_weight/std":0.0004983842458893556,"train/train/tensor_param_model_layers_22_input_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_83_mlp_gate_proj_weight/max_abs":0.001007080078125,"train/train/tensor_act_model_layers_24_self_attn_v_proj/norm":1291.8999545069307,"train/train/tensor_act_model_layers_51_mlp_up_proj/std":0.22388372081487415,"train/train/tensor_act_model_layers_30_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_24_self_attn_v_proj/std":0.2229070402954832,"train/train/tensor_act_model_layers_72_self_attn_v_proj/mean":-0.00994873046875,"train/train/tensor_act_model_layers_54_mlp_up_proj/mean":0.00782012939453125,"train/train/tensor_grad_model_layers_25_self_attn_v_proj_weight/max_abs":0.00616455078125,"train/train/tensor_act_model_layers_65_mlp_down_proj/std":0.016236471372754274,"train/train/layer_model_layers_33/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_mlp_down_proj/std":0.015213348939213412,"train/train/tensor_param_model_layers_16_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_23_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_self_attn/max_abs":0.23828125,"train/train/tensor_act_model_layers_86_mlp_up_proj/std":0.22583259912159825,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_63_mlp_down_proj/max_abs":0.08154296875,"train/train/tensor_act_model_layers_0_self_attn/std":0.00869760555080741,"train/train/tensor_act_model_layers_22_self_attn_o_proj/mean":6.896257400512695e-05,"train/train/layer_model_layers_63/act/max_abs":4.75,"train/train/tensor_param_model_layers_49_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_input_layernorm/max_abs":4.53125,"train/train/tensor_act_model_layers_22_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_7_input_layernorm/std":0.9961046031285845,"train/train/tensor_param_model_layers_68_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_14_post_attention_layernorm/std":1.0000033732446758,"train/train/tensor_act_model_layers_56_mlp/mean":-0.00043392181396484375,"train/train/tensor_act_model_layers_34_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_82_self_attn_q_proj/mean":0.00589752197265625,"train/train/layer_model_layers_87/act/max_abs":4.84375,"train/train/tensor_act_model_layers_6_mlp_up_proj/norm":1805.3100868528434,"train/train/tensor_act_model_layers_87_mlp_gate_proj/max_abs":1.1171875,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_grad_model_layers_35_mlp_down_proj_weight/std":0.00011372489154170928,"train/train/tensor_act_model_layers_42_self_attn/std":0.04931716459432309,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/norm":0.02879624182780685,"train/train/tensor_act_model_layers_54_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_28_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_11_self_attn_k_proj/mean":0.0015694797039031982,"train/train/tensor_grad_model_layers_65_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_12_mlp_up_proj/max_abs":1.25,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_65_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50/max_abs":1.734375,"train/train/tensor_grad_model_layers_49_mlp_up_proj_weight/norm":0.02978553146376114,"train/train/tensor_param_model_layers_13_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_47_self_attn_k_proj/norm":1318.2933468244928,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_48_input_layernorm_weight/max_abs":0.0004425048828125,"train/train/tensor_grad_model_layers_67_self_attn_o_proj_weight/std":0.0004342331585034654,"train/train/layer_model_layers_82/grad/norm":0.14765927785177926,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/mean":-8.866190910339355e-07,"train/train/tensor_act_model_layers_77_mlp/mean":0.0003638267517089844,"train/train/tensor_grad_model_layers_22_mlp_gate_proj_weight/norm":0.04826426136168903,"train/train/tensor_act_model_layers_85_self_attn/max_abs":0.25,"train/train/tensor_act_model_layers_72_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_66_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_31_self_attn_q_proj/mean":0.005077362060546875,"train/train/tensor_act_model_layers_14_self_attn/std":0.048890890866537265,"train/train/tensor_grad_model_layers_54_self_attn_v_proj_weight/max_abs":0.00518798828125,"train/train/tensor_param_model_layers_55_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_mlp/norm":89.99802737476747,"train/train/tensor_param_model_layers_68_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_0_self_attn_v_proj_weight/std":0.002998788371438506,"train/train/tensor_grad_model_layers_68_mlp_up_proj_weight/mean":-1.2523378245532513e-07,"train/train/tensor_grad_model_layers_16_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_63_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_25_mlp_up_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_40_self_attn_o_proj_weight/std":0.020263671875,"train/train/tensor_grad_model_layers_70_mlp_down_proj_weight/max_abs":0.0016937255859375,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_1_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_5_mlp_up_proj_weight/mean":-1.7369166016578674e-06,"train/train/tensor_act_model_layers_33_mlp/mean":-0.00017189979553222656,"train/train/layer_model_layers_62/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_56_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_75_self_attn_q_proj_weight/std":4.4215754747487545e-07,"train/train/tensor_act_model_layers_86_mlp_down_proj/norm":90.91961238020785,"train/train/tensor_param_model_layers_73_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_55_input_layernorm_weight/std":0.00010129027137051493,"train/train/tensor_param_model_layers_70_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_87_self_attn/norm":288.78435134613994,"train/train/tensor_param_model_layers_27_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_52_self_attn_o_proj_weight/max_abs":0.095703125,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/mean":1.8198988982476294e-09,"train/train/tensor_param_model_layers_33_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_46_self_attn/mean":0.0013294219970703125,"train/train/tensor_grad_model_layers_68_self_attn_k_proj_weight/norm":0.00011006962377921674,"train/train/layer_model_layers_23/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_49_self_attn_q_proj/frac_near_user_limit":0,"train/train/global/grad/max_abs":0.1259765625,"train/train/tensor_grad_model_layers_34_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/mean":-0.0001392364501953125,"train/train/layer_model_layers_29/grad/std":0.0002687315647936677,"train/train/tensor_param_model_layers_25_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_6_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_25_mlp_down_proj/std":0.0151829737139752,"train/train/tensor_param_model_layers_57_self_attn_o_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_1_mlp_up_proj/norm":1842.4492060691748,"train/train/tensor_act_model_layers_41_post_attention_layernorm/std":1.0000032396003282,"train/train/tensor_act_model_layers_53_mlp_down_proj/max_abs":0.08984375,"train/train/tensor_grad_model_layers_56_mlp_down_proj_weight/std":7.644111870043366e-05,"train/train/tensor_param_model_layers_23_input_layernorm_weight/frac_near_user_limit":0,"train/train/layer__model_layers_3/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_61_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_74_self_attn_o_proj/mean":-6.699562072753906e-05,"train/train/tensor_param_model_layers_54_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_64_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_86_self_attn/std":0.048525337544329854,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/mean":0.0003070831298828125,"train/train/tensor_act_model_layers_21_mlp/mean":-0.0003205537796020508,"train/train/tensor_param_model_layers_27_mlp_gate_proj_weight/max_abs":0.08349609375,"train/train/tensor_grad_model_layers_5_input_layernorm_weight/norm":0.009922202848289399,"train/train/layer_model_layers_64/act/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_23_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_34_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_76/max_abs":2.15625,"train/train/tensor_act_model_layers_20_mlp_gate_proj/mean":-0.01129150390625,"train/train/tensor_act_model_layers_66_post_attention_layernorm/mean":-0.0025911331176757812,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/max_abs":4.410743713378906e-06,"train/train/tensor_grad_model_layers_28_self_attn_q_proj_weight/max_abs":6.407499313354492e-06,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/mean":3.5762786865234375e-05,"train/train/tensor_act_model_layers_32_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/layer_model_layers_52/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_1_self_attn_v_proj_weight/std":0.002798466811508756,"train/train/tensor_param_model_layers_57_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_56/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_6_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83/std":0.45752492927114397,"train/train/tensor_grad_model_layers_47_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_70_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_64_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_45_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_89_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_81_mlp_up_proj_weight/norm":0.026070801014448582,"train/train/tensor_grad_model_layers_78_mlp_down_proj_weight/max_abs":0.00098419189453125,"train/train/tensor_grad_model_layers_9_input_layernorm_weight/mean":-2.002716064453125e-05,"train/train/layer_model_layers_9/grad/norm":0.4586588910602147,"train/train/tensor_param_model_layers_3_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_8_self_attn_q_proj_weight/norm":0.0004961911655180312,"train/train/tensor_act_model_layers_48_mlp/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_18_mlp_up_proj_weight/norm":3.640625,"train/train/tensor_act_model_layers_15_self_attn/max_abs":0.2412109375,"train/train/tensor_grad_model_layers_58_input_layernorm_weight/max_abs":0.000324249267578125,"train/train/tensor_act_model_layers_32_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/max_abs":0.005523681640625,"train/train/tensor_act_model_layers_27_input_layernorm/max_abs":4.6875,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_act_model_layers_56_input_layernorm/mean":-0.006214141845703125,"train/train/tensor_act_model_layers_72_post_attention_layernorm/mean":0.0010166168212890625,"train/train/tensor_grad_model_layers_16_post_attention_layernorm_weight/std":8.58911744115022e-05,"train/train/tensor_act_model_layers_8_mlp_down_proj/max_abs":0.09375,"train/train/tensor_param_model_layers_53_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_input_layernorm_weight/norm":0.0031366555586419976,"train/train/tensor_grad_model_layers_20_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_86_input_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_5_self_attn_o_proj/norm":262.9886154951609,"train/train/tensor_act_model_layers_18_mlp/norm":91.11790881556891,"train/train/tensor_param_model_layers_76_self_attn_k_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_63_self_attn_k_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_87_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/mean":3.1739473342895508e-06,"train/train/tensor_act_model_layers_32_mlp_gate_proj/mean":0.010894775390625,"train/train/tensor_grad_model_layers_42_self_attn_o_proj_weight/norm":0.12298437075056966,"train/train/tensor_grad_model_layers_14_mlp_gate_proj_weight/std":0.00018482247362471149,"train/train/tensor_act_model_layers_34_mlp_gate_proj/norm":1849.9072922770215,"train/train/tensor_act_model_layers_74_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_26_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_81_self_attn_o_proj_weight/std":0.00036504558818559527,"train/train/tensor_param_model_layers_82_self_attn_o_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_52_input_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_60_mlp_up_proj/std":0.22656550819409566,"train/train/tensor_act_model_layers_82/max_abs":2.125,"train/train/tensor_param_model_layers_49_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_act_model_layers_88_mlp_down_proj/norm":89.73871702312833,"train/train/layer_model_layers_73/grad/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_43_input_layernorm/norm":5792.589477539962,"train/train/layer_model_layers_33/act/max_abs":5.1875,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_36/grad/max_abs":0.007110595703125,"train/train/tensor_grad_model_layers_86_self_attn_q_proj_weight/std":3.7250225470998487e-07,"train/train/tensor_param_model_layers_57_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_31/act/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_o_proj_weight/std":0.0004367255771938048,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_q_proj/norm":1328.3761619836957,"train/train/tensor_act_model_layers_5_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_36_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_84_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_56_self_attn_q_proj/std":0.2255960974118705,"train/train/tensor_act_model_layers_67_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_60_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_self_attn/mean":0.003162384033203125,"train/train/tensor_act_model_layers_78_mlp_down_proj/max_abs":0.08154296875,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp_up_proj/frac_near_dtype_limit":0,"_timestamp":1.7862590146837573e+09,"train/train/tensor_act_model_layers_37_self_attn/std":0.04828006581961424,"train/train/tensor_act_model_layers_60_mlp_gate_proj/mean":0.0025997161865234375,"train/train/tensor_act_model_layers_8_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_0_post_attention_layernorm/norm":5786.478271522039,"train/train/tensor_param_model_layers_32_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_21_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81/mean":0.004085540771484375,"train/train/tensor_act_model_layers_8_post_attention_layernorm/max_abs":5.03125,"train/train/tensor_act_model_layers_59_self_attn_v_proj/norm":1330.4489675236166,"train/train/tensor_act_model_layers_58_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_45_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_61_post_attention_layernorm_weight/mean":7.309019565582275e-06,"train/train/tensor_param_model_layers_75_self_attn_k_proj_weight/mean":-0.00021266937255859375,"train/train/tensor_act_model_layers_72_self_attn/mean":0.0003476142883300781,"train/train/tensor_act_model_layers_20_self_attn_q_proj/max_abs":1.1796875,"train/train/tensor_param_model_layers_93_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_8_mlp_up_proj/std":0.2282760884845734,"train/train/tensor_grad_model_layers_72_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_78_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_80_mlp_up_proj_weight/max_abs":0.0771484375,"train/train/tensor_act_model_layers_18/norm":1199.4172018770143,"train/train/tensor_grad_model_layers_20_mlp_down_proj_weight/max_abs":0.002471923828125,"train/train/layer_model_layers_39/grad/std":0.0002420307402630765,"train/train/layer__model_layers_25/param/std":0.04421362932966808,"train/train/tensor_param_model_layers_26_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_70_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_5_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_15_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_input_layernorm_weight/mean":4.131346940994263e-06,"train/train/tensor_grad_model_layers_52_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_post_attention_layernorm_weight/max_abs":0.0002346038818359375,"train/train/tensor_param_model_layers_57_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_6_mlp_down_proj_weight/max_abs":0.003143310546875,"train/train/tensor_act_model_layers_82_self_attn_o_proj/max_abs":0.2412109375,"train/train/tensor_act_model_layers_53_mlp_gate_proj/mean":-0.00487518310546875,"train/train/tensor_param_model_layers_87_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_23_mlp_gate_proj/std":0.22461645734546812,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/norm":0.0954992497199979,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/max_abs":0.07666015625,"train/train/tensor_grad_model_layers_0_input_layernorm_weight/std":0.0006447778092047124,"train/train/tensor_act_model_layers_30_self_attn_k_proj/norm":1285.5651712293325,"train/train/tensor_param_model_layers_69_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_76_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_90_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_14/grad/max_abs":0.0091552734375,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/norm":2.546875,"train/train/tensor_grad_model_layers_83_self_attn_o_proj_weight/mean":2.889428287744522e-07,"train/train/tensor_act_model_layers_49/norm":1989.7092802822904,"train/train/tensor_grad_model_layers_47_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_4_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_66_self_attn_k_proj/norm":1323.6494803847716,"train/train/tensor_grad_model_layers_25_mlp_gate_proj_weight/std":0.00011725252093971867,"train/train/tensor_act_model_layers_39_self_attn_q_proj/max_abs":1.078125,"train/train/layer_model_layers_75/grad/frac_near_dtype_limit":0,"train/train/layer__model_layers_82/param/frac_near_dtype_limit":0,"train/train/layer_model_layers_23/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_74_mlp_down_proj/std":0.015777797024298185,"train/train/tensor_grad_model_layers_69_mlp_up_proj_weight/max_abs":0.000911712646484375,"train/train/tensor_act_model_layers_90_self_attn_k_proj/max_abs":1.09375,"train/train/tensor_act_model_layers_23_self_attn_o_proj/mean":0.00222015380859375,"train/train/tensor_grad_model_layers_64_post_attention_layernorm_weight/std":3.6955237783064195e-05,"train/train/tensor_act_model_layers_34_mlp_down_proj/frac_near_user_limit":0,"train/train/layer_model_layers_33/act/frac_near_user_limit":0,"train/train/tensor_param_model_layers_28_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_79_self_attn/mean":0.0006890296936035156,"train/train/tensor_param_model_layers_1_input_layernorm_weight/norm":11.3125,"train/train/tensor_act_model_layers_4/mean":-0.00439453125,"train/train/tensor_grad_model_layers_25_post_attention_layernorm_weight/max_abs":0.0001811981201171875,"train/train/tensor_param_model_layers_73_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/mean":8.254573913291097e-09,"train/train/tensor_act_model_layers_82/mean":0.0055999755859375,"train/train/tensor_act_model_layers_4_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_19_mlp_down_proj/max_abs":0.08203125,"train/train/tensor_act_model_layers_56_mlp/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/max_abs":0.00124359130859375,"train/train/tensor_act_model_layers_15_mlp_gate_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_12_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/norm":0.15306974698687836,"train/train/layer_model_layers_23/act/mean":-0.007068634033203125,"train/train/tensor_act_model_layers_4_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/norm":9.808889215001914e-05,"train/train/tensor_grad_model_layers_43_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_82_self_attn_k_proj_weight/mean":2.3756001610308886e-09,"train/train/tensor_act_model_layers_13_self_attn_q_proj/mean":-0.01141357421875,"train/train/tensor_grad_model_layers_35_self_attn_o_proj_weight/mean":6.146728992462158e-06,"train/train/tensor_act_model_layers_1_self_attn_o_proj/std":0.018647465918248556,"train/train/tensor_grad_model_layers_57_self_attn_o_proj_weight/max_abs":0.0036163330078125,"train/train/tensor_grad_model_layers_72_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_65_self_attn_k_proj_weight/max_abs":7.152557373046875e-06,"train/train/tensor_param_model_layers_29_self_attn_q_proj_weight/max_abs":0.091796875,"train/train/tensor_act_model_layers_2_mlp/std":0.015213279036291365,"train/train/tensor_act_model_layers_59_input_layernorm/norm":5792.588623049803,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/mean":-3.218650817871094e-05,"train/train/tensor_act_model_layers_54_post_attention_layernorm/std":1.0000102366669956,"train/train/tensor_grad_model_layers_40_self_attn_o_proj_weight/std":0.00045439931036915044,"train/train/tensor_act_model_layers_53_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_62_mlp_up_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_48_self_attn_k_proj/mean":-0.00015878677368164062,"train/train/tensor_act_model_layers_89_mlp_up_proj/max_abs":1.125,"train/train/tensor_param_model_layers_63_self_attn_q_proj_weight/max_abs":0.0751953125,"train/train/tensor_grad_model_layers_56_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_43_self_attn_o_proj_weight/std":0.00045396637859254136,"train/train/tensor_param_model_layers_8_self_attn_v_proj_weight/norm":2.578125,"train/train/tensor_param_model_layers_82_mlp_up_proj_weight/mean":-9.72747802734375e-05,"train/train/tensor_grad_model_layers_36_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_81_input_layernorm_weight/mean":1,"train/train/tensor_param_model_layers_61_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_56_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_85_mlp_gate_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/max_abs":0.00150299072265625,"train/train/tensor_act_model_layers_86_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_83_mlp_up_proj_weight/norm":0.029088295082042075,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/max_abs":0.00055694580078125,"train/train/layer_model_layers_12/act/std":0.4135001152765549,"train/train/tensor_grad_model_layers_11_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_15/act/mean":-0.007959485054016113,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/mean":-1.1303200153633952e-08,"train/train/layer__model_layers_84/param/std":0.044251178436925734,"train/train/tensor_grad_model_layers_80_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_20_mlp_gate_proj/max_abs":1.2421875,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/std":0.0201416015625,"train/train/tensor_act_model_layers_56_self_attn_o_proj/std":0.04760935711442432,"train/train/tensor_grad_model_layers_30_self_attn_o_proj_weight/std":0.0006236274915847308,"train/train/tensor_act_model_layers_55_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_83/max_abs":2.21875,"train/train/tensor_act_model_layers_28_mlp_down_proj/norm":89.53416260285036,"train/train/tensor_param_model_layers_70_self_attn_k_proj_weight/mean":0.00017547607421875,"train/train/tensor_grad_model_layers_22_self_attn_q_proj_weight/max_abs":8.463859558105469e-06,"train/train/tensor_param_model_layers_51_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_13_mlp_down_proj/max_abs":0.08056640625,"train/train/tensor_grad_model_layers_48_post_attention_layernorm_weight/norm":0.001072330332195051,"train/train/tensor_act_model_layers_23_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_37_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_10_mlp_gate_proj/norm":1839.5629164830007,"train/train/tensor_act_model_layers_4_self_attn_v_proj/std":0.22681071551864093,"train/train/tensor_act_model_layers_67_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_30_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_79/param/max_abs":1,"train/train/tensor_act_model_layers_34_self_attn_o_proj/max_abs":0.251953125,"train/train/tensor_param_model_layers_86_mlp_gate_proj_weight/norm":3.640625,"train/train/tensor_param_model_layers_61_mlp_up_proj_weight/max_abs":0.08154296875,"train/train/tensor_param_model_layers_69_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_34/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_57_mlp_up_proj_weight/norm":0.02909533798932028,"train/train/tensor_grad_model_layers_38_self_attn_o_proj_weight/std":0.0005756820819766964,"train/train/tensor_act_model_layers_16_self_attn_o_proj/mean":0.00031375885009765625,"train/train/tensor_grad_model_layers_7_self_attn_k_proj_weight/norm":0.000560844830250546,"train/train/tensor_act_model_layers_34_post_attention_layernorm/max_abs":5.03125,"train/train/tensor_act_model_layers_19_self_attn/std":0.049502737274082956,"train/train/layer_model_layers_14/grad/norm":0.3804425844875772,"train/train/tensor_param_model_layers_8_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_10_self_attn_q_proj/max_abs":1.234375,"train/train/tensor_act_model_layers_50_mlp_gate_proj/mean":0.002750396728515625,"train/train/tensor_grad_model_layers_29_self_attn_k_proj_weight/max_abs":8.285045623779297e-06,"train/train/tensor_grad_model_layers_45_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_26/grad/norm":0.25600465524728844,"train/train/layer_model_layers_53/grad/frac_near_user_limit":0,"train/train/tensor_act_model_layers_21_input_layernorm/norm":5792.5454101568985,"train/train/tensor_act_model_layers_5_self_attn/max_abs":0.2099609375,"train/train/layer__model_layers_81/param/mean":0.0016082192359960024,"train/train/tensor_param_model_layers_27_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_82_mlp_down_proj_weight/max_abs":0.0947265625,"train/train/tensor_param_model_layers_80_input_layernorm_weight/std":0,"train/train/tensor_act_model_layers_53_self_attn_q_proj/norm":1333.253961007645,"train/train/tensor_act_model_layers_76_self_attn_q_proj/mean":0.00594329833984375,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/mean":0.0001964569091796875,"train/train/tensor_grad_model_layers_1_mlp_down_proj_weight/std":0.0005879281875265551,"train/train/tensor_param_model_layers_49_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_84/grad/mean":5.407774337464232e-08,"train/train/tensor_param_model_layers_81_self_attn_v_proj_weight/max_abs":0.080078125,"train/train/tensor_grad_model_layers_1_self_attn_o_proj_weight/norm":0.8757451835236343,"train/train/tensor_act_model_layers_13_mlp_up_proj/norm":1851.440568919327,"train/train/tensor_param_model_layers_85_mlp_gate_proj_weight/max_abs":0.07958984375,"train/train/tensor_grad_model_layers_62_mlp_gate_proj_weight/std":7.74981719192448e-05,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/norm":0.00010540959673920654,"train/train/tensor_act_model_layers_34_self_attn_q_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_15_mlp_down_proj/max_abs":0.0830078125,"train/train/tensor_grad_model_layers_71_self_attn_k_proj_weight/mean":2.172100721509196e-09,"train/train/tensor_param_model_layers_74_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_71_mlp_up_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_6_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_20_self_attn_k_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_0_mlp_up_proj/norm":1858.5692695825799,"train/train/layer_model_layers_2/grad/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/norm":0.09826652574240087,"train/train/tensor_grad_model_layers_78_mlp_gate_proj_weight/std":8.255359438517036e-05,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/mean":1,"train/train/tensor_grad_model_layers_90_self_attn_o_proj_weight/mean":3.306195139884949e-08,"train/train/tensor_grad_model_layers_28_mlp_down_proj_weight/std":0.00011457428692613792,"train/train/tensor_grad_model_layers_53_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_92_mlp_gate_proj/norm":1845.2338920906745,"train/train/tensor_grad_model_layers_37_input_layernorm_weight/norm":0.002752573049069252,"train/train/tensor_act_model_layers_33_self_attn_o_proj/norm":273.46574315856384,"train/train/tensor_grad_model_layers_71_mlp_down_proj_weight/mean":-4.249159246683121e-08,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/mean":1.8408172763884068e-09,"train/train/tensor_param_model_layers_17_input_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_12_post_attention_layernorm/std":1.0000053197003966,"train/train/tensor_act_model_layers_79_mlp_up_proj/mean":0.002246856689453125,"train/train/tensor_act_model_layers_36_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/mean":-8.678436279296875e-05,"train/train/tensor_grad_model_layers_68_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/norm":0.00012173270687591402,"train/train/tensor_param_model_layers_91_input_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_22_mlp_gate_proj_weight/mean":1.319684088230133e-06,"train/train/tensor_grad_model_layers_66_mlp_up_proj_weight/std":7.976348733026974e-05,"train/train/tensor_param_model_layers_46_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_45_mlp_down_proj/mean":0.00023669004440307617,"train/train/tensor_act_model_layers_35_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_7_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/norm":2.546875,"train/train/tensor_act_model_layers_91_self_attn/std":0.04699854976164658,"train/train/tensor_param_model_layers_33_self_attn_q_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_3_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_55_mlp/mean":-0.0002148151397705078,"train/train/tensor_act_model_layers_5_self_attn_v_proj/max_abs":1.203125,"train/train/tensor_act_model_layers_27_self_attn_q_proj/std":0.22778739297995734,"train/train/layer_model_layers_80/act/mean":0.0039384620530264714,"train/train/tensor_grad_model_layers_24_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/layer_model_layers_34/act/mean":-0.002246856689453125,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/max_abs":0.0036773681640625,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/mean":-0.0002651214599609375,"train/train/layer_model_layers_15/act/norm":8983.979974151278,"train/train/tensor_param_model_layers_14_mlp_down_proj_weight/mean":-5.269050598144531e-05,"train/train/tensor_grad_model_layers_65_input_layernorm_weight/norm":0.0022154427059947633,"train/train/tensor_grad_model_layers_16_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_52_mlp_up_proj_weight/std":8.919218010107343e-05,"train/train/tensor_act_model_layers_19_mlp_gate_proj/max_abs":1.0859375,"train/train/tensor_grad_model_layers_79_self_attn_o_proj_weight/max_abs":0.004425048828125,"train/train/tensor_act_model_layers_5_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_mlp/max_abs":0.07958984375,"train/train/tensor_param_model_layers_67_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_57_mlp_up_proj/std":0.21997337612238516,"train/train/tensor_grad_model_layers_60_self_attn_q_proj_weight/max_abs":5.8710575103759766e-06,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/std":0.0004229783738890793,"train/train/tensor_act_model_layers_75_self_attn_k_proj/std":0.23511166175097137,"train/train/tensor_grad_model_layers_51_input_layernorm_weight/max_abs":0.000804901123046875,"train/train/layer_model_layers_65/act/mean":0.0022890163319451468,"train/train/tensor_grad_model_layers_89_mlp_up_proj_weight/norm":0.02458262427702942,"train/train/tensor_act_model_layers_73_post_attention_layernorm/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_93_self_attn_v_proj_weight/norm":2.59375,"train/train/tensor_act_model_layers_25_self_attn_k_proj/std":0.22192974733153176,"train/train/tensor_act_model_layers_0/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_42_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_rotary_emb/mean":0.333984375,"train/train/tensor_param_model_layers_25_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_mlp_gate_proj_weight/mean":3.409385681152344e-05,"train/train/tensor_param_model_layers_20_mlp_up_proj_weight/mean":-3.147125244140625e-05,"train/train/tensor_param_model_layers_17_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_6_self_attn_q_proj/std":0.21582720603117567,"train/train/tensor_param_model_layers_26_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_35_mlp_gate_proj_weight/mean":2.42842361330986e-07,"train/train/tensor_act_model_layers_4_mlp_gate_proj/mean":-0.0097198486328125,"train/train/tensor_grad_model_layers_43_mlp_down_proj_weight/std":9.490587959794029e-05,"train/train/tensor_act_model_layers_74_post_attention_layernorm/norm":5792.596801758938,"train/train/tensor_grad_model_layers_68_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_29_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_11_input_layernorm_weight/mean":1,"train/train/layer__model_layers_54/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_26_self_attn_k_proj/mean":0.0128326416015625,"train/train/tensor_param_model_layers_25_mlp_gate_proj_weight/max_abs":0.08740234375,"train/train/tensor_act_model_layers_81_mlp_down_proj/std":0.015426764818718871,"train/train/tensor_param_model_layers_20_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_mlp_up_proj/norm":1848.397558926767,"train/train/tensor_act_model_layers_93_self_attn_q_proj/norm":1346.0201985565452,"train/train/tensor_act_model_layers_61_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_87_mlp_gate_proj_weight/mean":1.1431438906583935e-08,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/norm":0.11863149451916977,"train/train/tensor_act_model_layers_83_mlp_down_proj/max_abs":0.08056640625,"train/train/tensor_grad_model_layers_38_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_2_mlp_down_proj/norm":88.23609882394013,"train/train/layer_model_layers_32/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_10_input_layernorm_weight/max_abs":0.00109100341796875,"train/train/tensor_param_model_layers_22_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_47_self_attn_o_proj_weight/mean":9.441375732421875e-05,"train/train/tensor_grad_model_layers_52_self_attn_v_proj_weight/mean":4.002358764410019e-07,"train/train/layer__model_layers_10/param/std":0.04421595299596808,"train/train/layer__model_layers_84/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_55_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model/max_abs":4.375,"train/train/layer_model_layers_64/grad/max_abs":0.005523681640625,"train/train/tensor_param_model_layers_61_mlp_gate_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_46_self_attn_o_proj_weight/std":0.00047884427213594786,"train/train/tensor_act_model_layers_79_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_88_mlp_down_proj_weight/std":6.763570601355187e-05,"train/train/tensor_act_model_layers_28_mlp_up_proj/max_abs":1.171875,"train/train/tensor_grad_model_layers_86_self_attn_v_proj_weight/std":0.0004086024987981393,"train/train/layer__model_layers_56/param/max_abs":1,"train/train/tensor_act_model_layers_79_mlp_gate_proj/mean":0.003536224365234375,"train/train/tensor_act_model_layers_45_input_layernorm/mean":-0.007114410400390625,"train/train/tensor_act_model_layers_17_self_attn/std":0.046936715169399845,"train/train/tensor_grad_model_layers_49_self_attn_k_proj_weight/std":5.720800353433399e-07,"train/train/tensor_param_model_layers_73_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_86_self_attn_k_proj/std":0.22437876985905283,"train/train/tensor_grad_model_layers_23_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_88/param/norm":17.937302644820235,"train/train/tensor_act_model_layers_73_mlp_gate_proj/max_abs":1.3203125,"train/train/tensor_param_model_layers_58_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_41_mlp_down_proj_weight/std":9.877160883794701e-05,"train/train/tensor_grad_model_layers_34_self_attn_v_proj_weight/std":0.0006212403831783952,"train/train/tensor_param_model_layers_41_post_attention_layernorm_weight/norm":11.3125,"train/train/tensor_grad_model_layers_32_input_layernorm_weight/mean":3.423541784286499e-06,"train/train/tensor_act_model_layers_5_self_attn/mean":-0.0015811920166015625,"train/train/layer_model_layers_48/act/mean":-0.0006498864718845912,"train/train/tensor_param_model_layers_90_self_attn_q_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_14_self_attn_v_proj_weight/max_abs":0.08251953125,"train/train/tensor_grad_model_layers_67_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_49_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_7_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_88_self_attn_v_proj/mean":-0.0110626220703125,"train/train/tensor_act_model_layers_67_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_70_self_attn_o_proj_weight/std":0.0201416015625,"train/train/tensor_param_model_layers_33_mlp_gate_proj_weight/mean":7.390975952148438e-05,"train/train/tensor_act_model_layers_32_mlp/frac_near_user_limit":0,"train/train/global/param/max_abs":1,"train/train/tensor_act_model_layers_70_self_attn_k_proj/norm":1325.4288855143154,"train/train/tensor_grad_model_layers_22_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_65_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/norm":0.04696170417491087,"train/train/tensor_param_model_layers_32_post_attention_layernorm_weight/std":0,"train/train/tensor_act_model_layers_59_mlp_up_proj/std":0.2307141799640801,"train/train/layer__model_layers_2/param/frac_near_user_limit":0,"train/train/tensor_act_model_layers_11_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_act_model_layers_50_self_attn/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_46_self_attn_o_proj/mean":0.0013294219970703125,"train/train/tensor_grad_model_layers_11_input_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_14_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_39_mlp_down_proj_weight/max_abs":0.001434326171875,"train/train/tensor_act_model_layers_80_self_attn/norm":285.57978478466794,"train/train/layer__model_layers_90/param/frac_near_user_limit":0,"train/train/tensor_param_model_layers_54_mlp_up_proj_weight/mean":0.000133514404296875,"train/train/layer_model_layers_62/act/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_12_self_attn_o_proj/mean":0.0014553070068359375,"train/train/tensor_grad_model_layers_0_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/mean":0.00014019012451171875,"train/train/tensor_act_model_layers_83_mlp/frac_near_user_limit":0,"train/train/tensor_act_model_layers_80_self_attn_k_proj/max_abs":0.98828125,"train/train/tensor_act_model_layers_3_mlp_down_proj/norm":88.42887849898709,"train/train/tensor_act_model_layers_68_post_attention_layernorm/mean":0.0043239593505859375,"train/train/tensor_grad_model_layers_51_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_66_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_90/std":0.4804788542742408,"train/train/tensor_param_model_layers_77_mlp_down_proj_weight/norm":3.609375,"train/train/layer__model_layers_16/param/frac_near_user_limit":0,"train/train/layer_model_layers_54/act/mean":-0.00040265492030552456,"train/train/tensor_act_model_layers_93_mlp/mean":-0.0004558563232421875,"train/train/layer__model_layers_82/param/std":0.04424587631498612,"train/train/tensor_grad_model_layers_73_mlp_down_proj_weight/std":7.439806445204105e-05,"train/train/tensor_act_model_layers_81/norm":2631.6569195902653,"train/train/tensor_act_model_layers_6_self_attn_v_proj/norm":1320.0946784996197,"train/train/tensor_grad_model_layers_16_input_layernorm_weight/std":0.00022461964474650948,"train/train/tensor_act_model_layers_64_self_attn_k_proj/max_abs":1.078125,"train/train/tensor_act_model_layers_9_self_attn_k_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_84_mlp_gate_proj/max_abs":1.1171875,"train/train/tensor_act_model_layers_56_input_layernorm/norm":5792.593750000895,"train/train/tensor_grad_model_layers_75_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_36_mlp_up_proj_weight/std":0.00010988061748079294,"train/train/tensor_act_model_layers_54_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_73_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_30_mlp_up_proj_weight/norm":0.0417089446471107,"train/train/tensor_act_model_layers_49_mlp_up_proj/std":0.22364313922177215,"train/train/tensor_grad_model_layers_21_self_attn_q_proj_weight/std":9.87389095151142e-07,"train/train/tensor_act_model_layers_51_self_attn_v_proj/norm":1330.4864214794868,"train/train/tensor_param_model_layers_51_mlp_up_proj_weight/norm":3.625,"train/train/tensor_param_model_layers_63_post_attention_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_61_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_42_self_attn_k_proj_weight/std":5.750666427769156e-07,"train/train/tensor_param_model_layers_75_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/mean":-0.00012969970703125,"train/train/tensor_grad_model_layers_92_self_attn_v_proj_weight/std":0.00034716524131420216,"train/train/tensor_act_model_layers_63_self_attn/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/std":5.742679981112172e-05,"train/train/tensor_grad_model_layers_63_self_attn_o_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_1_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_7/act/norm":8943.161235494905,"train/train/tensor_act_model_layers_51_mlp/frac_near_user_limit":0,"train/train/tensor_param_model_layers_0_post_attention_layernorm_weight/mean":1,"train/train/tensor_act_model_layers_15_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_41_mlp_down_proj_weight/norm":3.625,"train/train/tensor_act_model_layers_27_mlp/max_abs":0.08154296875,"train/train/tensor_act_model_layers_72_mlp/norm":89.93018348327459,"train/train/tensor_act_model_layers_0_post_attention_layernorm/mean":-0.0391845703125,"train/train/tensor_act_model_layers_55_self_attn_o_proj/norm":284.8669635533363,"train/train/tensor_act_model_layers_28_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_56_self_attn_v_proj_weight/max_abs":0.08837890625,"train/train/tensor_act_model_layers_13_self_attn_q_proj/std":0.22803587560600502,"train/train/tensor_act_model_layers_51_mlp/norm":90.24203654185172,"train/train/tensor_param_model_norm_weight/max_abs":1,"train/train/tensor_act_model_layers_66_self_attn/max_abs":0.2412109375,"train/train/tensor_act_model_layers_38_self_attn_o_proj/mean":-0.003543853759765625,"train/train/tensor_act_model_layers_79_mlp_up_proj/max_abs":1.1953125,"train/train/tensor_act_model_layers_85_self_attn_v_proj/max_abs":1.015625,"train/train/tensor_act_model_layers_43_input_layernorm/mean":-0.01119232177734375,"train/train/tensor_act_model_layers_39_self_attn/norm":288.09591313449556,"train/train/tensor_grad_model_layers_19_mlp_up_proj_weight/std":0.0001681896159259047,"train/train/tensor_act_model_layers_28_mlp_up_proj/mean":-0.00507354736328125,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/mean":1.3707904145121574e-08,"train/train/layer_model_layers_83/grad/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_81_input_layernorm/max_abs":4.75,"train/train/tensor_grad_model_layers_48_mlp_down_proj_weight/norm":0.030552653183268618,"train/train/tensor_act_model_layers_26_self_attn_o_proj/max_abs":0.2392578125,"train/train/tensor_act_model_layers_49_self_attn_o_proj/max_abs":0.2353515625,"train/train/tensor_act_model_layers_93_mlp_down_proj/mean":-0.0004558563232421875,"train/train/tensor_grad_model_layers_16_mlp_down_proj_weight/std":0.00017293972350399255,"train/train/tensor_act_model_layers_10_mlp_down_proj/norm":89.99802737476747,"train/train/tensor_act_model_layers_33_mlp_down_proj/mean":-0.00017189979553222656,"train/train/tensor_act_model_layers_23_self_attn/norm":282.9412194841048,"train/train/tensor_param_model_layers_72_self_attn_o_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_58_self_attn_o_proj_weight/max_abs":0.0791015625,"train/train/tensor_grad_model_layers_23_mlp_down_proj_weight/max_abs":0.00162506103515625,"train/train/tensor_grad_model_layers_80_mlp_down_proj_weight/max_abs":0.00101470947265625,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/std":0.00012568340413311694,"train/train/tensor_act_model_layers_69_self_attn_o_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_31_post_attention_layernorm_weight/std":0,"train/train/tensor_param_model_layers_16_input_layernorm_weight/mean":1,"train/train/layer__model_layers_59/param/max_abs":1,"train/train/tensor_act_model_layers_89_self_attn_o_proj/norm":274.62135541157573,"train/train/tensor_act_model_layers_4_mlp_down_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/norm":0.2348424700044126,"train/train/tensor_grad_model_layers_41_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_93_self_attn_o_proj_weight/norm":2.59375,"train/train/tensor_act_model_layers_3_post_attention_layernorm/std":1.0000207398730043,"train/train/tensor_act_model_layers_90_input_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_14_mlp_gate_proj_weight/mean":-4.076957702636719e-05,"train/train/tensor_grad_model_layers_49_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_89_mlp_gate_proj/mean":0.00970458984375,"train/train/tensor_grad_model_layers_24_self_attn_o_proj_weight/std":0.0006752105694234271,"train/train/tensor_param_model_layers_33_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_79_mlp_down_proj_weight/mean":-0.00011777877807617188,"train/train/tensor_param_model_layers_16_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_2_self_attn_k_proj_weight/norm":0.002385390930769212,"train/train/tensor_act_model_layers_81_post_attention_layernorm/std":1.0000008130442772,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/norm":0.0038242449486782914,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/mean":-1.6748905181884766e-05,"train/train/tensor_param_model_layers_19_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_72_mlp_gate_proj_weight/std":7.323531073077715e-05,"train/train/layer__model_layers_79/param/norm":17.92782023228842,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_grad_model_layers_88_mlp_gate_proj_weight/std":6.499570979273745e-05,"train/train/tensor_grad_model_layers_12_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_73_self_attn/frac_near_user_limit":0,"train/train/tensor_param_model_layers_42_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_56_input_layernorm_weight/std":0,"train/train/tensor_grad_model_layers_46_mlp_up_proj_weight/std":8.96853573035788e-05,"train/train/tensor_param_model_layers_48_input_layernorm_weight/std":0,"train/train/tensor_param_model_layers_76_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/layer__model_layers_74/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_25_mlp_gate_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_1_input_layernorm/std":1.000000897794559,"train/train/tensor_act_model_layers_29_mlp_gate_proj/std":0.22192436839389137,"train/train/tensor_act_model_layers_37_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_35_self_attn_o_proj/norm":266.3064247910867,"train/train/layer_model_layers_74/grad/mean":1.7680236474028243e-09,"train/train/tensor_param_model_layers_87_mlp_gate_proj_weight/std":0.0198974609375,"train/train/tensor_grad_model_layers_58_self_attn_v_proj_weight/std":0.00040723102992410424,"train/train/tensor_param_model_layers_29_mlp_down_proj_weight/mean":0.0001430511474609375,"train/train/tensor_act_model_layers_83_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_34_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_32_mlp_gate_proj_weight/max_abs":0.0018768310546875,"train/train/tensor_act_model_layers_28_mlp_gate_proj/norm":1861.4817607957123,"train/train/tensor_param_model_layers_55_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_39_input_layernorm/std":1.0000053880358246,"train/train/layer_model_layers_73/grad/max_abs":0.004974365234375,"train/train/tensor_act_model_layers_20_self_attn/frac_near_user_limit":0,"train/train/tensor_act_model_layers_47/std":0.33838399230680993,"train/train/tensor_grad_model_layers_37_self_attn_v_proj_weight/std":0.0005596112090983617,"train/train/layer_model_layers_6/grad/max_abs":0.0169677734375,"train/train/tensor_act_model_layers_47_self_attn_q_proj/norm":1343.373249612319,"train/train/tensor_param_model_layers_40_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_62_mlp_up_proj_weight/norm":0.02765680668157463,"train/train/tensor_act_model_layers_58_self_attn_k_proj/mean":-0.001926422119140625,"train/train/layer_model_layers_1/grad/std":0.0014777557953378696,"train/train/tensor_param_model_layers_39_mlp_up_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_40_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/layer__model_layers_72/param/norm":17.937302644820235,"train/train/tensor_param_model_layers_18_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_18_post_attention_layernorm_weight/norm":0.0015508797053401155,"train/train/tensor_grad_model_layers_30_mlp_down_proj_weight/mean":5.80446794629097e-07,"train/train/tensor_act_model_layers_60_self_attn_v_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_76_self_attn/max_abs":0.251953125,"train/train/tensor_param_model_layers_38_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/layer_model_layers_93/act/std":0.43135691795807746,"train/train/tensor_act_model_layers_77_self_attn_k_proj/mean":-0.010955810546875,"train/train/tensor_param_model_layers_59_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/mean":3.420282155275345e-07,"train/train/tensor_act_model_layers_39/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_71_self_attn_o_proj_weight/norm":0.11075000230776266,"train/train/tensor_grad_model_layers_60_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_12_mlp_down_proj_weight/mean":9.965896606445312e-05,"train/train/tensor_act_model_layers_43_self_attn_o_proj/std":0.047671234442833124,"train/train/tensor_act_model_layers_30_self_attn_q_proj/max_abs":1.09375,"train/train/tensor_grad_model_layers_31_input_layernorm_weight/max_abs":0.000888824462890625,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/norm":2.5625,"train/train/tensor_param_model_layers_21_mlp_down_proj_weight/norm":3.609375,"train/train/tensor_param_model_layers_77_self_attn_k_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_36_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_33_mlp/frac_near_user_limit":0,"train/train/layer_model_layers_6/act/mean":-0.011165193149021693,"train/train/tensor_act_model_layers_41_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_86_mlp_up_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_10_post_attention_layernorm_weight/std":9.742046900636952e-05,"train/train/tensor_param_model_layers_41_self_attn_v_proj_weight/max_abs":0.0810546875,"train/train/tensor_grad_model_layers_6_self_attn_v_proj_weight/std":0.0016303804843836433,"train/train/tensor_act_model_layers_23_mlp/norm":89.8131948203714,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/std":0.0201416015625,"train/train/tensor_grad_model_layers_58_mlp_down_proj_weight/max_abs":0.00138092041015625,"train/train/tensor_act_model_layers_32_post_attention_layernorm/std":1.0000011548393257,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/norm":0.00014712029868943168,"train/train/tensor_act_model_layers_87/std":0.47022663479972243,"train/train/tensor_param_model_layers_41_self_attn_k_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_0_mlp/norm":90.1570072565903,"train/train/tensor_grad_model_layers_21_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/layer_model_layers_12/act/mean":-0.007098139928919929,"train/train/tensor_grad_model_layers_30_self_attn_k_proj_weight/mean":-4.179128154646605e-10,"train/train/tensor_act_model_layers_89_self_attn_v_proj/max_abs":1.078125,"train/train/tensor_act_model_layers_56_self_attn_k_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_33_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_8_mlp_up_proj_weight/mean":4.410743713378906e-05,"train/train/tensor_act_model_layers_90_self_attn_v_proj/norm":1344.6113805768841,"train/train/tensor_act_model_layers_48_post_attention_layernorm/max_abs":4.71875,"train/train/tensor_param_model_layers_1_self_attn_v_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_40_mlp_up_proj_weight/max_abs":0.0986328125,"train/train/tensor_act_model_layers_92_mlp_down_proj/norm":88.58318316199377,"train/train/tensor_grad_model_layers_43_self_attn_q_proj_weight/std":5.383596239190834e-07,"train/train/tensor_param_model_layers_40_self_attn_v_proj_weight/std":0.0198974609375,"train/train/tensor_act_model_layers_78_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_77_self_attn_q_proj/max_abs":1.0859375,"train/train/tensor_param_model_layers_49_mlp_gate_proj_weight/mean":-7.009506225585938e-05,"train/train/layer_model_layers_82/grad/mean":1.2251153002041224e-07,"train/train/tensor_act_model_layers_52_self_attn_v_proj/norm":1288.9974399537014,"train/train/layer_model_layers_61/grad/mean":-6.587899474546243e-08,"train/train/tensor_param_model_layers_37_self_attn_v_proj_weight/max_abs":0.07958984375,"train/train/tensor_param_model_layers_24_input_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_44_self_attn_k_proj/norm":1287.1436957413978,"train/train/tensor_param_model_layers_15_self_attn_v_proj_weight/max_abs":0.0810546875,"train/train/tensor_act_model_layers_65/mean":0.0008554458618164062,"train/train/layer_model_layers_18/grad/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_74_post_attention_layernorm_weight/frac_near_user_limit":0,"train/train/tensor_act_model_layers_68_self_attn/max_abs":0.23828125,"train/train/tensor_act_model_layers_80_mlp/norm":90.12947819630477,"train/train/tensor_param_model_layers_70_mlp_down_proj_weight/norm":3.625,"train/train/layer_model_layers_16/act/max_abs":4.71875,"train/train/tensor_act_model_layers_72_post_attention_layernorm/frac_near_user_limit":0,"train/train/tensor_param_model_layers_7_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_param_model_layers_42_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_9_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_2_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_6_mlp_down_proj_weight/max_abs":0.083984375,"train/train/tensor_grad_model_layers_4_self_attn_v_proj_weight/mean":-1.1868774890899658e-05,"train/train/tensor_grad_model_layers_18_self_attn_o_proj_weight/max_abs":0.00787353515625,"train/train/tensor_act_model_layers_57_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_54/max_abs":1.78125,"train/train/tensor_grad_model_layers_20_self_attn_o_proj_weight/std":0.0009173559099532179,"train/train/tensor_grad_model_layers_68_post_attention_layernorm_weight/norm":0.0007899463455341934,"train/train/tensor_grad_model_layers_50_self_attn_q_proj_weight/max_abs":6.467103958129883e-06,"train/train/tensor_param_model_layers_38_mlp_down_proj_weight/mean":4.363059997558594e-05,"train/train/tensor_act_model_layers_10_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_73_mlp_gate_proj_weight/norm":0.028967146264284593,"train/train/tensor_act_model_layers_73_mlp/norm":88.27401653108261,"train/train/tensor_act_model_layers_58_self_attn_o_proj/std":0.050539023130170065,"train/train/tensor_grad_model_layers_59_mlp_gate_proj_weight/max_abs":0.00103759765625,"train/train/tensor_param_model_layers_9_input_layernorm_weight/max_abs":1,"train/train/layer_model_layers_19/grad/mean":-3.5815787619319794e-06,"train/train/tensor_param_model_layers_60_self_attn_o_proj_weight/mean":-4.076957702636719e-05,"train/train/tensor_param_model_layers_22_self_attn_o_proj_weight/max_abs":0.06982421875,"train/train/layer__model_layers_35/param/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_6_input_layernorm_weight/max_abs":1,"train/train/tensor_param_model_layers_64_input_layernorm_weight/max_abs":1,"train/train/tensor_act_model_layers_9_mlp/max_abs":0.08544921875,"train/train/tensor_act_model_layers_82_post_attention_layernorm/norm":5792.594848634802,"train/train/tensor_grad_model_layers_57_self_attn_q_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_73_self_attn/std":0.04889374723608988,"train/train/tensor_param_model_layers_21_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_93_self_attn_q_proj_weight/frac_near_user_limit":0,"train/train/tensor_param_model_layers_27_self_attn_k_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_0_mlp_up_proj_weight/max_abs":0.00836181640625,"train/train/tensor_act_model_layers_69_self_attn/norm":281.2734756313297,"train/train/tensor_act_model_layers_11_self_attn_k_proj/std":0.21729071365931837,"train/train/tensor_param_model_layers_70_post_attention_layernorm_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_79_mlp_down_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_20_post_attention_layernorm_weight/norm":0.0014519872335425547,"train/train/tensor_act_model_layers_27_mlp_up_proj/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_16_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_param_model_layers_87_self_attn_k_proj_weight/max_abs":0.08935546875,"train/train/tensor_param_model_layers_50_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_52_self_attn_v_proj/std":0.22266361405822618,"train/train/tensor_param_model_layers_62_self_attn_k_proj_weight/max_abs":0.0859375,"train/train/tensor_act_model_layers_83_mlp_down_proj/std":0.015732491614499727,"train/train/tensor_param_model_layers_0_self_attn_o_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_64_self_attn_k_proj_weight/max_abs":5.334615707397461e-06,"train/train/tensor_act_model_layers_7_self_attn_v_proj/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_28_mlp_gate_proj_weight/frac_near_user_limit":0,"train/train/tensor_grad_model_layers_27_post_attention_layernorm_weight/norm":0.0013541638001005082,"train/train/tensor_act_model_layers_34_mlp/frac_near_dtype_limit":0,"train/train/tensor_act_model_layers_43_self_attn_o_proj/mean":-0.0018444061279296875,"train/train/tensor_param_model_layers_67_self_attn_v_proj_weight/std":0.02001953125,"train/train/tensor_grad_model_layers_69_mlp_gate_proj_weight/norm":0.027074234913612157,"train/train/tensor_param_model_layers_79_self_attn_q_proj_weight/max_abs":0.07861328125,"train/train/tensor_act_model_layers_93_mlp_down_proj/frac_near_dtype_limit":0,"train/train/tensor_param_model_layers_82_self_attn_v_proj_weight/frac_near_dtype_limit":0,"train/train/tensor_grad_model_layers_28_input_layernorm_weight/max_abs":0.000553131103515625,"train/train/layer_model_layers_20/act/mean":-0.006148099899291992,"train/train/tensor_act_model_layers_41_self_attn_q_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_90_input_layernorm/std":1.000004251944175,"train/train/tensor_param_model_layers_21_self_attn_v_proj_weight/mean":7.677078247070312e-05,"train/train/tensor_param_model_layers_8_mlp_down_proj_weight/mean":-0.00021457672119140625,"train/train/tensor_act_model_layers_15_mlp/norm":94.01638228446947,"train/train/tensor_act_model_layers_88_self_attn_k_proj/max_abs":1.21875,"train/train/tensor_grad_model_layers_33_self_attn_q_proj_weight/max_abs":9.059906005859375e-06,"train/train/tensor_act_model_layers_87_input_layernorm/std":1.0000039416986133,"train/train/tensor_param_model_layers_47_mlp_down_proj_weight/std":0.02001953125,"train/train/tensor_act_model_layers_41_mlp/norm":90.19993339537602,"train/train/tensor_grad_model_layers_9_self_attn_k_proj_weight/norm":0.0003711668094756757,"train/train/tensor_act_model_layers_81_mlp_gate_proj/frac_near_user_limit":0,"train/train/tensor_act_model_layers_3_self_attn_k_proj/mean":-0.01470947265625} \ No newline at end of file diff --git a/wandb/run-20260809_070213-uvqyddz0/logs/debug-core.log b/wandb/run-20260809_070213-uvqyddz0/logs/debug-core.log new file mode 100644 index 0000000000000000000000000000000000000000..6ab6c95a5b7718c427082e9937cc663568192adb --- /dev/null +++ b/wandb/run-20260809_070213-uvqyddz0/logs/debug-core.log @@ -0,0 +1,58 @@ +{"time":"2026-08-09T03:56:53.721154509Z","level":"INFO","msg":"main: starting server","port-filename":"/tmp/tmpyqomfhgp/port-2869678.txt","pid":2869678,"detached":false,"idle-timeout":600000000000,"log-level":0,"disable-analytics":false,"shutdown-on-parent-exit":false,"enable-dcgm-profiling":false} +{"time":"2026-08-09T03:56:53.722258921Z","level":"INFO","msg":"server: will exit if parent process dies","ppid":2869678} +{"time":"2026-08-09T03:56:53.722234577Z","level":"INFO","msg":"server: accepting connections","addr":{"Name":"/tmp/wandb-2869678-2909401-2521299324/socket","Net":"unix"}} +{"time":"2026-08-09T03:56:53.900147851Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"1(@)"} +{"time":"2026-08-09T03:58:19.204299995Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"2(@)"} +{"time":"2026-08-09T03:58:19.282614427Z","level":"INFO","msg":"handleInformInit: received","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T03:58:19.544211049Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T03:58:24.897517362Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"xv8tfjtqu29x"} +{"time":"2026-08-09T05:00:38.67636271Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"xv8tfjtqu29x"} +{"time":"2026-08-09T05:00:40.520576799Z","level":"INFO","msg":"connection: cancelling request","id":"2(@)","requestId":"xv8tfjtqu29x"} +{"time":"2026-08-09T05:00:40.552184373Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T05:00:40.553143275Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"cvzjg5ej","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608126464Z","level":"INFO","msg":"processOutgoingData: finished","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608106419Z","level":"INFO","msg":"connection: closing","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608216462Z","level":"INFO","msg":"connection: closed successfully","id":"2(@)"} +{"time":"2026-08-09T05:00:42.608222262Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"2(@)"} +{"time":"2026-08-09T05:00:50.241711721Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"3(@)"} +{"time":"2026-08-09T05:00:50.316234404Z","level":"INFO","msg":"handleInformInit: received","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:00:50.57530289Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:00:55.905488223Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"4jxnuihlia82"} +{"time":"2026-08-09T05:57:13.765576248Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"4jxnuihlia82"} +{"time":"2026-08-09T05:57:15.666017984Z","level":"INFO","msg":"connection: cancelling request","id":"3(@)","requestId":"4jxnuihlia82"} +{"time":"2026-08-09T05:57:15.999725293Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:57:16.001240777Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"59pftr14","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068849513Z","level":"INFO","msg":"connection: closing","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068936758Z","level":"INFO","msg":"connection: closed successfully","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068853914Z","level":"INFO","msg":"processOutgoingData: finished","id":"3(@)"} +{"time":"2026-08-09T05:57:18.068948948Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"3(@)"} +{"time":"2026-08-09T05:57:26.287713024Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"4(@)"} +{"time":"2026-08-09T05:57:26.365087395Z","level":"INFO","msg":"handleInformInit: received","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T05:57:26.623707494Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T05:57:31.962348796Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"rz2ldpq16nht"} +{"time":"2026-08-09T07:02:02.153459338Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"rz2ldpq16nht"} +{"time":"2026-08-09T07:02:04.144735717Z","level":"INFO","msg":"connection: cancelling request","id":"4(@)","requestId":"rz2ldpq16nht"} +{"time":"2026-08-09T07:02:04.180963791Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T07:02:04.181861619Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"m1dnjnh6","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233097548Z","level":"INFO","msg":"connection: closing","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233183816Z","level":"INFO","msg":"connection: closed successfully","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233105065Z","level":"INFO","msg":"processOutgoingData: finished","id":"4(@)"} +{"time":"2026-08-09T07:02:06.233192773Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"4(@)"} +{"time":"2026-08-09T07:02:13.885910885Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"5(@)"} +{"time":"2026-08-09T07:02:13.954769181Z","level":"INFO","msg":"handleInformInit: received","streamId":"uvqyddz0","id":"5(@)"} +{"time":"2026-08-09T07:02:14.215272022Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"uvqyddz0","id":"5(@)"} +{"time":"2026-08-09T07:02:19.530617395Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"fk71ydt1h8f7"} +{"time":"2026-08-09T07:03:39.9336748Z","level":"INFO","msg":"connection: cancelling request","id":"5(@)","requestId":"fk71ydt1h8f7"} +{"time":"2026-08-09T07:03:40.29901782Z","level":"INFO","msg":"connection: closing","id":"5(@)"} +{"time":"2026-08-09T07:03:40.299111085Z","level":"INFO","msg":"connection: closed successfully","id":"5(@)"} +{"time":"2026-08-09T07:03:40.299007456Z","level":"INFO","msg":"processOutgoingData: finished","id":"5(@)"} +{"time":"2026-08-09T07:03:40.299122073Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"5(@)"} +{"time":"2026-08-09T07:03:42.349833633Z","level":"INFO","msg":"connection: closing","id":"1(@)"} +{"time":"2026-08-09T07:03:42.349942779Z","level":"INFO","msg":"connection: closed successfully","id":"1(@)"} +{"time":"2026-08-09T07:03:42.34985531Z","level":"INFO","msg":"processOutgoingData: finished","id":"1(@)"} +{"time":"2026-08-09T07:03:42.349954994Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"1(@)"} +{"time":"2026-08-09T07:03:42.353940558Z","level":"INFO","msg":"server: parent process exited, terminating service process"} +{"time":"2026-08-09T07:03:42.353988499Z","level":"INFO","msg":"server: is shutting down"} +{"time":"2026-08-09T07:03:42.354128892Z","level":"INFO","msg":"server: listener closed","addr":{"Name":"/tmp/wandb-2869678-2909401-2521299324/socket","Net":"unix"}} +{"time":"2026-08-09T07:03:42.354189401Z","level":"INFO","msg":"server: forced shutdown"} +{"time":"2026-08-09T07:03:42.354198506Z","level":"ERROR","msg":"main: Serve() returned error","error":"forced shutdown"} diff --git a/wandb/run-20260809_070213-uvqyddz0/logs/debug-internal.log b/wandb/run-20260809_070213-uvqyddz0/logs/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..04886cdd8d75e6ef4a4955a1434a422418e44d50 --- /dev/null +++ b/wandb/run-20260809_070213-uvqyddz0/logs/debug-internal.log @@ -0,0 +1,40 @@ +{"time":"2026-08-09T07:02:13.954953014Z","level":"INFO","msg":"wandb-core"} +{"time":"2026-08-09T07:02:13.955105829Z","level":"INFO","msg":"stream: starting","core version":"0.28.1"} +{"time":"2026-08-09T07:02:14.215123169Z","level":"INFO","msg":"stream: created new stream","id":"uvqyddz0"} +{"time":"2026-08-09T07:02:14.215184619Z","level":"INFO","msg":"handler: started"} +{"time":"2026-08-09T07:02:14.215266968Z","level":"INFO","msg":"stream: started"} +{"time":"2026-08-09T07:02:14.215275401Z","level":"INFO","msg":"writer: started","stream_id":"uvqyddz0"} +{"time":"2026-08-09T07:02:14.215296287Z","level":"INFO","msg":"sender: started"} +{"time":"2026-08-09T07:02:15.114109279Z","level":"INFO","msg":"filestream: sending request","total_files":1,"console_offset":0,"console_lines":1} +{"time":"2026-08-09T07:02:15.211087322Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:02:30.114358029Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":0,"events_lines":2,"console_offset":1,"console_lines":4,"uploaded_len":2} +{"time":"2026-08-09T07:02:30.230090173Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:02:45.11459263Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":2,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T07:02:45.228934504Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:02:56.598899602Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":737} +{"time":"2026-08-09T07:02:56.598941378Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1} +{"time":"2026-08-09T07:02:56.606194892Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":3095} +{"time":"2026-08-09T07:02:56.606379301Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":14} +{"time":"2026-08-09T07:02:56.610759144Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":4507} +{"time":"2026-08-09T07:02:56.610912588Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":13} +{"time":"2026-08-09T07:02:56.613175719Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":5301} +{"time":"2026-08-09T07:02:56.618404048Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1932} +{"time":"2026-08-09T07:02:56.629287315Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":10255} +{"time":"2026-08-09T07:02:56.629322845Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":1} +{"time":"2026-08-09T07:02:56.63322527Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":11868} +{"time":"2026-08-09T07:02:56.633338533Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":10} +{"time":"2026-08-09T07:02:56.636529541Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":13176} +{"time":"2026-08-09T07:02:56.63797994Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":477} +{"time":"2026-08-09T07:02:56.643113777Z","level":"INFO","msg":"flowcontrol: backed up, offloading to disk","recordNumber":14979} +{"time":"2026-08-09T07:02:56.643188638Z","level":"INFO","msg":"flowcontrol: unblocked","totalOffloaded":12} +{"time":"2026-08-09T07:03:00.155597928Z","level":"INFO","msg":"filestream: sending request","total_files":4,"history_offset":0,"history_lines":1,"events_offset":4,"events_lines":2,"console_offset":4,"console_lines":2} +{"time":"2026-08-09T07:03:01.144833138Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:03:15.114830048Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":6,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T07:03:15.243621253Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:03:30.114703175Z","level":"INFO","msg":"filestream: sending request","total_files":2,"events_offset":8,"events_lines":2,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T07:03:30.219274044Z","level":"INFO","msg":"filestream: request sent","status":"200 OK"} +{"time":"2026-08-09T07:03:40.29903607Z","level":"ERROR","msg":"runupserter: failed to upload changes","error":"POST https://api.wandb.ai/graphql giving up after 1 attempt(s): context canceled"} +{"time":"2026-08-09T07:03:40.30029433Z","level":"ERROR","msg":"runfiles: CreateRunFiles returned error: POST https://api.wandb.ai/graphql giving up after 1 attempt(s): context canceled"} +{"time":"2026-08-09T07:03:40.300493766Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"} +{"time":"2026-08-09T07:03:40.322418888Z","level":"INFO","msg":"filestream: sending request","total_files":3,"history_offset":1,"history_lines":1,"console_offset":4,"console_lines":1} +{"time":"2026-08-09T07:03:40.3225757Z","level":"ERROR+4","msg":"filestream: fatal error: filestream: error making HTTP request: POST https://api.wandb.ai/files/deepnevro-deepnevro/huggingface/uvqyddz0/file_stream giving up after 1 attempt(s): context canceled. got response: "} diff --git a/wandb/run-20260809_070213-uvqyddz0/logs/debug.log b/wandb/run-20260809_070213-uvqyddz0/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..6b23e39286dbd88cb2f9a763f55c9b5f9e2fb13b --- /dev/null +++ b/wandb/run-20260809_070213-uvqyddz0/logs/debug.log @@ -0,0 +1,27 @@ +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_setup.py:_flush():81] Current SDK version is 0.28.1 +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_setup.py:_flush():81] Configure stats pid to 474098 +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_setup.py:_flush():81] Loading settings from environment variables +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_init.py:setup_run_log_directory():729] Logging user logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_070213-uvqyddz0/logs/debug.log +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_init.py:setup_run_log_directory():730] Logging internal logs to /mnt/data/zainulabideen/zain-exp/notebooks/Activation/wandb/run-20260809_070213-uvqyddz0/logs/debug-internal.log +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_init.py:init():772] calling init triggers +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_init.py:init():777] wandb.init called with sweep_config: {} +config: {'_wandb': {}} +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_init.py:init():820] starting backend +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_init.py:init():826] Connected to an existing wandb-core service via WANDB_SERVICE +2026-08-09 07:02:13,953 INFO MainThread:474098 [wandb_init.py:init():835] sending inform_init request +2026-08-09 07:02:14,215 INFO MainThread:474098 [wandb_init.py:init():840] backend started and connected +2026-08-09 07:02:14,218 INFO MainThread:474098 [wandb_init.py:init():910] updated telemetry +2026-08-09 07:02:14,225 INFO MainThread:474098 [wandb_init.py:init():933] communicating run to backend with 90.0 second timeout +2026-08-09 07:02:14,452 INFO MainThread:474098 [wandb_init.py:init():978] starting run threads in backend +2026-08-09 07:02:14,525 INFO MainThread:474098 [wandb_run.py:_console_start():2621] atexit reg +2026-08-09 07:02:14,525 INFO MainThread:474098 [wandb_run.py:_redirect():2471] redirect: wrap_raw +2026-08-09 07:02:14,525 INFO MainThread:474098 [wandb_run.py:_redirect():2540] Wrapping output streams. +2026-08-09 07:02:14,525 INFO MainThread:474098 [wandb_run.py:_redirect():2563] Redirects installed. +2026-08-09 07:02:14,528 INFO MainThread:474098 [wandb_init.py:init():1016] run started, returning control to user process +2026-08-09 07:02:14,529 INFO MainThread:474098 [wandb_run.py:_config_callback():1346] config_cb None None {'transformers_version': '5.15.0.dev0', 'architectures': None, 'output_hidden_states': False, 'return_dict': True, 'dtype': None, 'chunk_size_feed_forward': 0, 'is_encoder_decoder': False, 'id2label': {0: 'LABEL_0', 1: 'LABEL_1'}, 'label2id': {'LABEL_0': 0, 'LABEL_1': 1}, 'problem_type': None, 'vocab_size': 4096, 'hidden_size': 128, 'intermediate_size': 256, 'num_hidden_layers': 94, 'num_attention_heads': 4, 'num_key_value_heads': 4, 'hidden_act': 'silu', 'max_position_embeddings': 512, 'initializer_range': 0.02, 'rms_norm_eps': 1e-06, 'use_cache': False, 'pad_token_id': 0, 'bos_token_id': 1, 'eos_token_id': 2, 'pretraining_tp': 1, 'tie_word_embeddings': True, 'rope_parameters': {'rope_theta': 10000.0, 'rope_type': 'default'}, 'attention_bias': False, 'attention_dropout': 0.0, 'mlp_bias': False, 'head_dim': 32, '_name_or_path': '', 'tokenizer_name': 'w-ahmad/tiny-stories-tokenizer', 'mlp_type': 'glu', 'activation': 'tanh', 'model_type': 'tiny_llama', 'output_attentions': False, 'output_dir': 'out/glu-tanh-94L_run', 'per_device_train_batch_size': 128, 'num_train_epochs': 1, 'max_steps': 1500, 'learning_rate': 0.001, 'lr_scheduler_type': 'constant', 'lr_scheduler_kwargs': None, 'warmup_steps': 0, 'optim': 'adamw_torch_fused', 'optim_args': None, 'weight_decay': 0.01, 'adam_beta1': 0.9, 'adam_beta2': 0.999, 'adam_epsilon': 1e-08, 'optim_target_modules': None, 'gradient_accumulation_steps': 4, 'average_tokens_across_devices': True, 'max_grad_norm': 1.0, 'label_smoothing_factor': 0.0, 'bf16': True, 'fp16': False, 'bf16_full_eval': False, 'fp16_full_eval': False, 'tf32': None, 'gradient_checkpointing': False, 'gradient_checkpointing_kwargs': None, 'torch_compile': False, 'torch_compile_backend': None, 'torch_compile_mode': None, 'use_liger_kernel': False, 'liger_kernel_config': None, 'neftune_noise_alpha': None, 'torch_empty_cache_steps': None, 'auto_find_batch_size': False, 'logging_strategy': 'steps', 'logging_steps': 20, 'logging_first_step': False, 'log_on_each_node': True, 'logging_nan_inf_filter': True, 'include_num_input_tokens_seen': 'no', 'log_level': 'passive', 'log_level_replica': 'warning', 'disable_tqdm': False, 'report_to': ['wandb'], 'run_name': 'LM-glu-tanh-94L-15.9M-20260809-070212', 'project': 'huggingface', 'trackio_space_id': None, 'trackio_bucket_id': None, 'trackio_static_space_id': None, 'eval_strategy': 'steps', 'eval_steps': 50, 'eval_delay': 0, 'per_device_eval_batch_size': 128, 'prediction_loss_only': False, 'eval_on_start': False, 'eval_do_concat_batches': True, 'eval_use_gather_object': False, 'eval_accumulation_steps': None, 'include_for_metrics': [], 'batch_eval_metrics': False, 'save_only_model': False, 'save_strategy': 'steps', 'save_steps': 100, 'save_on_each_node': False, 'save_total_limit': None, 'enable_jit_checkpoint': False, 'push_to_hub': True, 'hub_token': '', 'hub_private_repo': None, 'hub_model_id': 'w-ahmad/A-glu-tanh-94L', 'hub_strategy': 'every_save', 'hub_always_push': False, 'hub_revision': None, 'load_best_model_at_end': False, 'metric_for_best_model': None, 'greater_is_better': None, 'ignore_data_skip': False, 'restore_callback_states_from_checkpoint': False, 'full_determinism': False, 'seed': 42, 'data_seed': 42, 'use_cpu': False, 'accelerator_config': {'split_batches': False, 'dispatch_batches': None, 'even_batches': True, 'use_seedable_sampler': True, 'non_blocking': False, 'gradient_accumulation_kwargs': None}, 'parallelism_config': None, 'dataloader_drop_last': False, 'dataloader_num_workers': 0, 'dataloader_pin_memory': True, 'dataloader_persistent_workers': False, 'dataloader_prefetch_factor': None, 'dataloader_multiprocessing_context': None, 'dataloader_in_order': True, 'remove_unused_columns': False, 'label_names': None, 'train_sampling_strategy': 'random', 'length_column_name': 'length', 'ddp_find_unused_parameters': None, 'ddp_bucket_cap_mb': None, 'ddp_broadcast_buffers': None, 'ddp_static_graph': None, 'ddp_backend': None, 'ddp_timeout': 1800, 'fsdp': None, 'fsdp_config': None, 'deepspeed': None, 'debug': [], 'skip_memory_metrics': True, 'do_train': False, 'do_eval': True, 'do_predict': False, 'resume_from_checkpoint': None, 'local_rank': -1} +2026-08-09 07:02:14,532 INFO MainThread:474098 [wandb_config.py:__setitem__():155] [no run ID] config set model/num_parameters = 15949440 - > +2026-08-09 07:02:14,532 INFO MainThread:474098 [wandb_run.py:_config_callback():1346] config_cb model/num_parameters 15949440 None +2026-08-09 07:03:39,932 INFO MainThread:474098 [wandb_run.py:_finish():2383] finishing run deepnevro-deepnevro/huggingface/uvqyddz0 +2026-08-09 07:03:39,933 INFO MainThread:474098 [wandb_run.py:_atexit_cleanup():2588] got exitcode: 0 +2026-08-09 07:03:39,933 INFO MainThread:474098 [wandb_run.py:_restore():2570] restore +2026-08-09 07:03:39,933 INFO MainThread:474098 [wandb_run.py:_restore():2576] restore done diff --git a/wandb/run-20260809_070213-uvqyddz0/run-uvqyddz0.wandb b/wandb/run-20260809_070213-uvqyddz0/run-uvqyddz0.wandb new file mode 100644 index 0000000000000000000000000000000000000000..86e927cef225b25fa9cf8cac79fbbdeabe96a84a --- /dev/null +++ b/wandb/run-20260809_070213-uvqyddz0/run-uvqyddz0.wandb @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a07bb65786bb7d1dd6c0b884d55358d7dad50e8e2b7eb8cc1d841d13d0ceab87 +size 6717440