{ "architecture": "block", "batch_size": 32, "canonical_sha256_without_self": "a0d2c00339b4ed2c671f57d7a804a28f3934cfaeaa7e1412046d0ce71e823782", "depth": 16, "diagnostics": [ { "activation_grad_rms_by_block": [ 0.00027713447343558073, 0.00024001447309274226, 0.00020889485313091427, 0.00019106797117274255, 0.00017709653184283525, 0.00016345064796041697, 0.0001520135992905125, 0.00014396758342627436, 0.00013824697816744447, 0.0001333381951553747, 0.00012879272981081158, 0.00012578140012919903, 0.00012338913802523166, 0.0001212825154652819, 0.00011928759340662509, 0.00011723516217898577 ], "activation_grad_statistics": { "first_quartile_mean": 0.00022927794270799495, "first_to_last_ratio": 1.9059069547229377, "imbalance_abs_log_ratio": 0.6449579870010378, "last_quartile_mean": 0.00012029860226903111, "mean": 0.00016006211535568582, "normalized": [ 1.7314182860806244, 1.4995083162520277, 1.3050861702453052, 1.1937113960299495, 1.1064237870985023, 1.021170110098828, 0.9497162957812465, 0.8994482117542517, 0.8637083038683806, 0.8330403159990363, 0.8046421823465957, 0.7858286756343993, 0.7708828397715447, 0.7577215582573746, 0.7452581339534802, 0.7324354168284537 ], "population_cv": 0.28697332073733334 }, "activation_output_rms_by_block": [ 0.005004490725696087, 0.007150472607463598, 0.0056586493737995625, 0.007886149920523167, 0.005775847006589174, 0.008285986259579659, 0.006597128231078386, 0.009296777658164501, 0.007501539774239063, 0.01081777922809124, 0.007698386441916227, 0.011321330443024635, 0.009009703993797302, 0.013792803511023521, 0.00953624676913023, 0.012954792007803917 ], "activation_output_statistics": { "first_quartile_mean": 0.0064249406568706036, "first_to_last_ratio": 0.567404514268178, "imbalance_abs_log_ratio": 0.5666828005795043, "last_quartile_mean": 0.011323386570438743, "mean": 0.008643005246995017, "normalized": [ 0.5790220626599802, 0.8273132322752738, 0.6547085431617609, 0.9124314628022479, 0.6682683674867963, 0.9586927258271091, 0.7632910130850682, 1.0756417926966766, 0.8679318778438997, 1.251622429808468, 0.8907071350665657, 1.3098835554867694, 1.04242722714183, 1.5958342170183197, 1.103348487778111, 1.4988758698611242 ], "population_cv": 0.28641618387108014 }, "bits_per_byte": 8.092078157393424, "branch_output_rms_by_sublayer": [ 0.0033052791841328144, 0.003737553022801876, 0.0033692142460495234, 0.0038969451561570168, 0.004222586750984192, 0.003809630637988448, 0.004224879201501608, 0.003801452461630106, 0.004351162351667881, 0.0037875697016716003, 0.005037518683820963, 0.0037579576019197702, 0.005422801710665226, 0.003782435320317745, 0.005750134587287903, 0.003670723643153906, 0.006515308283269405, 0.003754077944904566, 0.006654165219515562, 0.0036852110642939806, 0.006751408334821463, 0.003720880253240466, 0.007695567328482866, 0.003703791182488203, 0.008140962570905685, 0.003749529365450144, 0.009426429867744446, 0.0036915375385433435, 0.008798080496490002, 0.003752112854272127, 0.008380400948226452, 0.0036098985001444817 ], "capture": { "all_gradients_finite": true, "all_gradients_present": true, "count": 16, "dtypes": [ "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32" ], "position": "post-MLP Transformer-block output; Block AttnRes is captured before aggregation-partial reset", "shape": [ 16, 256, 192 ], "storage_unique": true }, "core_parameter_grad_rms_by_block": [ 0.007836087849617155, 0.00707663314616961, 0.006860545956961424, 0.006477463491545069, 0.00630808320659222, 0.0065403273634858875, 0.006076989561376891, 0.0053044433268254945, 0.006002429844215364, 0.005811298045810947, 0.00541486468124986, 0.005552686577658202, 0.005724443119797102, 0.005942467552368739, 0.006312250195815329, 0.004848514440357715 ], "core_parameter_grad_statistics": { "first_quartile_mean": 0.007062682611073314, "first_to_last_ratio": 1.237564932158175, "imbalance_abs_log_ratio": 0.21314568451503188, "last_quartile_mean": 0.005706918827084721, "mean": 0.006130595522490438, "normalized": [ 1.2781935818257821, 1.154314147819503, 1.119066807097797, 1.0565798163950184, 1.0289511326347756, 1.0668339379905794, 0.9912559944760846, 0.865241118479574, 0.9790940899942118, 0.947917380047643, 0.883252640201938, 0.9057336366895933, 0.9337499267072272, 0.9693132633801169, 1.0296308364592772, 0.7908716898008789 ], "population_cv": 0.1166547247084935 }, "depth_weights": [ { "entropy_mean": 0.0, "mean_weights": [ 1.0 ], "sources": 1 }, { "entropy_mean": 0.6931471824645996, "mean_weights": [ 0.5, 0.5 ], "sources": 2 }, { "entropy_mean": 0.6931471824645996, "mean_weights": [ 0.5, 0.5 ], "sources": 2 }, { "entropy_mean": 0.6931471824645996, "mean_weights": [ 0.5, 0.5 ], "sources": 2 }, { "entropy_mean": 0.6931471824645996, "mean_weights": [ 0.5, 0.5 ], "sources": 2 }, { "entropy_mean": 1.0986123085021973, "mean_weights": [ 0.3333333432674408, 0.3333333432674408, 0.3333333432674408 ], "sources": 3 }, { "entropy_mean": 1.0986123085021973, "mean_weights": [ 0.3333333432674408, 0.3333333432674408, 0.3333333432674408 ], "sources": 3 }, { "entropy_mean": 1.0986123085021973, "mean_weights": [ 0.3333333432674408, 0.3333333432674408, 0.3333333432674408 ], "sources": 3 }, { "entropy_mean": 1.0986123085021973, "mean_weights": [ 0.3333333432674408, 0.3333333432674408, 0.3333333432674408 ], "sources": 3 }, { "entropy_mean": 1.3862943649291992, "mean_weights": [ 0.25, 0.25, 0.25, 0.25 ], "sources": 4 }, { "entropy_mean": 1.3862943649291992, "mean_weights": [ 0.25, 0.25, 0.25, 0.25 ], "sources": 4 }, { "entropy_mean": 1.3862943649291992, "mean_weights": [ 0.25, 0.25, 0.25, 0.25 ], "sources": 4 }, { "entropy_mean": 1.3862943649291992, "mean_weights": [ 0.25, 0.25, 0.25, 0.25 ], "sources": 4 }, { "entropy_mean": 1.6094379425048828, "mean_weights": [ 0.20000001788139343, 0.20000001788139343, 0.20000001788139343, 0.20000001788139343, 0.20000001788139343 ], "sources": 5 }, { "entropy_mean": 1.6094379425048828, "mean_weights": [ 0.20000001788139343, 0.20000001788139343, 0.20000001788139343, 0.20000001788139343, 0.20000001788139343 ], "sources": 5 }, { "entropy_mean": 1.6094379425048828, "mean_weights": [ 0.20000001788139343, 0.20000001788139343, 0.20000001788139343, 0.20000001788139343, 0.20000001788139343 ], "sources": 5 }, { "entropy_mean": 1.6094379425048828, "mean_weights": [ 0.20000001788139343, 0.20000001788139343, 0.20000001788139343, 0.20000001788139343, 0.20000001788139343 ], "sources": 5 }, { "entropy_mean": 1.791759729385376, "mean_weights": [ 0.1666666567325592, 0.1666666567325592, 0.1666666567325592, 0.1666666567325592, 0.1666666567325592, 0.1666666567325592 ], "sources": 6 }, { "entropy_mean": 1.791759729385376, "mean_weights": [ 0.1666666567325592, 0.1666666567325592, 0.1666666567325592, 0.1666666567325592, 0.1666666567325592, 0.1666666567325592 ], "sources": 6 }, { "entropy_mean": 1.791759729385376, "mean_weights": [ 0.1666666567325592, 0.1666666567325592, 0.1666666567325592, 0.1666666567325592, 0.1666666567325592, 0.1666666567325592 ], "sources": 6 }, { "entropy_mean": 1.791759729385376, "mean_weights": [ 0.1666666567325592, 0.1666666567325592, 0.1666666567325592, 0.1666666567325592, 0.1666666567325592, 0.1666666567325592 ], "sources": 6 }, { "entropy_mean": 1.9459102153778076, "mean_weights": [ 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548 ], "sources": 7 }, { "entropy_mean": 1.9459102153778076, "mean_weights": [ 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548 ], "sources": 7 }, { "entropy_mean": 1.9459102153778076, "mean_weights": [ 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548 ], "sources": 7 }, { "entropy_mean": 1.9459102153778076, "mean_weights": [ 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548, 0.1428571492433548 ], "sources": 7 }, { "entropy_mean": 2.079441547393799, "mean_weights": [ 0.125, 0.125, 0.125, 0.125, 0.125, 0.125, 0.125, 0.125 ], "sources": 8 }, { "entropy_mean": 2.079441547393799, "mean_weights": [ 0.125, 0.125, 0.125, 0.125, 0.125, 0.125, 0.125, 0.125 ], "sources": 8 }, { "entropy_mean": 2.079441547393799, "mean_weights": [ 0.125, 0.125, 0.125, 0.125, 0.125, 0.125, 0.125, 0.125 ], "sources": 8 }, { "entropy_mean": 2.079441547393799, "mean_weights": [ 0.125, 0.125, 0.125, 0.125, 0.125, 0.125, 0.125, 0.125 ], "sources": 8 }, { "entropy_mean": 2.1972246170043945, "mean_weights": [ 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519 ], "sources": 9 }, { "entropy_mean": 2.1972246170043945, "mean_weights": [ 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519 ], "sources": 9 }, { "entropy_mean": 2.1972246170043945, "mean_weights": [ 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519 ], "sources": 9 } ], "layer_input_rms_by_sublayer": [ 0.02823692187666893, 0.014202049933373928, 0.014330258592963219, 0.014431006275117397, 0.0145475585013628, 0.009805543348193169, 0.009861846454441547, 0.009975572116672993, 0.010043247602880001, 0.007605585735291243, 0.007654537446796894, 0.0077590313740074635, 0.007819762453436852, 0.006375181954354048, 0.006421996746212244, 0.0065260399132966995, 0.006563509814441204, 0.005602568853646517, 0.0056375316344201565, 0.0057973419316112995, 0.005820456892251968, 0.005056646186858416, 0.00509054446592927, 0.005272198468446732, 0.005295110866427422, 0.0046798414550721645, 0.004703808110207319, 0.004869809839874506, 0.004889858886599541, 0.004459173884242773, 0.00446688337251544, 0.00453129643574357 ], "loss_nats": 5.609001159667969, "loss_scale": 1.0, "output_weights": { "entropy_mean": 2.1972246170043945, "mean_weights": [ 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519, 0.11111113429069519 ], "sources": 9 }, "step": 0, "stream_state_rms_by_sublayer": [ 0.0033052791841328144, 0.005004490725696087, 0.005999269895255566, 0.007150472607463598, 0.004222586750984192, 0.0056586493737995625, 0.006890070624649525, 0.007886149920523167, 0.004351162351667881, 0.005775847006589174, 0.007411441300064325, 0.008285986259579659, 0.005422801710665226, 0.006597128231078386, 0.008578632958233356, 0.009296777658164501, 0.006515308283269405, 0.007501539774239063, 0.010256913490593433, 0.01081777922809124, 0.006751408334821463, 0.007698386441916227, 0.01081965770572424, 0.011321330443024635, 0.008140962570905685, 0.009009703993797302, 0.013233783654868603, 0.013792803511023521, 0.008798080496490002, 0.00953624676913023, 0.01231932733207941, 0.012954792007803917 ] }, { "activation_grad_rms_by_block": [ 5.571244764723815e-05, 5.411167876445688e-05, 5.283674909151159e-05, 5.185072586755268e-05, 5.127084659761749e-05, 5.077067544334568e-05, 5.0770395318977535e-05, 5.0520964578026906e-05, 5.0666069000726566e-05, 5.061549381935038e-05, 5.055486690253019e-05, 5.0508093409007415e-05, 4.999646989745088e-05, 5.004305057809688e-05, 5.0446498789824545e-05, 5.0497859774623066e-05 ], "activation_grad_statistics": { "first_quartile_mean": 5.362790034268983e-05, "first_to_last_ratio": 1.067307499464034, "imbalance_abs_log_ratio": 0.06513912148859854, "last_quartile_mean": 5.024596975999884e-05, "mean": 5.132330534252105e-05, "normalized": [ 1.0855194784401525, 1.0543295760732245, 1.0294884310137495, 1.010276433318387, 0.9989778767257201, 0.9892323790237741, 0.989226920989334, 0.9843669311799096, 0.9871941930199502, 0.9862087697110136, 0.985027495114306, 0.9841161451299155, 0.9741475059679974, 0.9750550991234875, 0.9829160155051417, 0.9839167496639366 ], "population_cv": 0.029989256005628663 }, "activation_output_rms_by_block": [ 0.005180664826184511, 0.007948367856442928, 0.0071101076900959015, 0.011628935113549232, 0.009834222495555878, 0.015592618845403194, 0.011937727220356464, 0.02099907398223877, 0.014203419908881187, 0.023388667032122612, 0.014353072270751, 0.022941188886761665, 0.013975659385323524, 0.02055426314473152, 0.015514259226620197, 0.022412775084376335 ], "activation_output_statistics": { "first_quartile_mean": 0.007967018871568143, "first_to_last_ratio": 0.4398207829260811, "imbalance_abs_log_ratio": 0.8213879465753375, "last_quartile_mean": 0.018114239210262895, "mean": 0.014848438935587183, "normalized": [ 0.3489029957060359, 0.5352998985902224, 0.47884546792694416, 0.7831756027684648, 0.6623068282273259, 1.0501183937950838, 0.8039718701839673, 1.4142277227480384, 0.9565598087782763, 1.575159997194527, 0.966638468394887, 1.5450236207510426, 0.9412207873130777, 1.3842709818787224, 1.04484109702854, 1.5094364587148448 ], "population_cv": 0.38149080499888866 }, "bits_per_byte": 6.756301290892851, "branch_output_rms_by_sublayer": [ 0.00344275776296854, 0.0037783386651426554, 0.003971979953348637, 0.003924910444766283, 0.0057159834541380405, 0.0038860586937516928, 0.007017855532467365, 0.003858900861814618, 0.008758665062487125, 0.0038810574915260077, 0.009854576550424099, 0.0037807892076671124, 0.010700107552111149, 0.003945234231650829, 0.012973167933523655, 0.003866987768560648, 0.012833327986299992, 0.0038173284847289324, 0.01477400679141283, 0.003928930964320898, 0.012848040089011192, 0.004507644101977348, 0.014026274904608727, 0.004126352723687887, 0.012859554030001163, 0.004076994024217129, 0.01296316273510456, 0.0034613758325576782, 0.01440997514873743, 0.0041273003444075584, 0.01337564829736948, 0.0038736232090741396 ], "capture": { "all_gradients_finite": true, "all_gradients_present": true, "count": 16, "dtypes": [ "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32" ], "position": "post-MLP Transformer-block output; Block AttnRes is captured before aggregation-partial reset", "shape": [ 16, 256, 192 ], "storage_unique": true }, "core_parameter_grad_rms_by_block": [ 0.0004508021297651076, 0.0004524930113583151, 0.0005201601896034384, 0.0005594057048027396, 0.0006552041707250334, 0.0007507040984739487, 0.000850349803788574, 0.0009063852334487553, 0.0010217915468968724, 0.0011277818818807989, 0.0010544629528892386, 0.0011012469919556404, 0.0011079217036188914, 0.001237384982851684, 0.0013584092718518022, 0.0014241647000383885 ], "core_parameter_grad_statistics": { "first_quartile_mean": 0.0004957152588824002, "first_to_last_ratio": 0.3866823679479825, "imbalance_abs_log_ratio": 0.950151677612245, "last_quartile_mean": 0.0012819701645901916, "mean": 0.0009111667733718268, "normalized": [ 0.494752599567352, 0.4966083318466913, 0.5708726490086493, 0.61394436359068, 0.7190825981290028, 0.8238931888351498, 0.9332537452411748, 0.9947522889740842, 1.1214100170879497, 1.2377337660232879, 1.1572666867418089, 1.208611885484378, 1.215937340997369, 1.3580225036879552, 1.490845925850575, 1.5630121089338918 ], "population_cv": 0.3380559322236514 }, "depth_weights": [ { "entropy_mean": 0.0, "mean_weights": [ 1.0 ], "sources": 1 }, { "entropy_mean": 0.6931420564651489, "mean_weights": [ 0.5014879107475281, 0.49851205945014954 ], "sources": 2 }, { "entropy_mean": 0.693139910697937, "mean_weights": [ 0.5018156170845032, 0.4981843829154968 ], "sources": 2 }, { "entropy_mean": 0.6931352019309998, "mean_weights": [ 0.5023940205574036, 0.49760597944259644 ], "sources": 2 }, { "entropy_mean": 0.6931431889533997, "mean_weights": [ 0.49870234727859497, 0.501297652721405 ], "sources": 2 }, { "entropy_mean": 1.0986039638519287, "mean_weights": [ 0.33161261677742004, 0.33409368991851807, 0.3342936635017395 ], "sources": 3 }, { "entropy_mean": 1.0986011028289795, "mean_weights": [ 0.3315771222114563, 0.33318889141082764, 0.3352339565753937 ], "sources": 3 }, { "entropy_mean": 1.0986087322235107, "mean_weights": [ 0.3331664800643921, 0.3342593312263489, 0.3325742185115814 ], "sources": 3 }, { "entropy_mean": 1.0985949039459229, "mean_weights": [ 0.3315378725528717, 0.3324793577194214, 0.3359827995300293 ], "sources": 3 }, { "entropy_mean": 1.3862826824188232, "mean_weights": [ 0.24934276938438416, 0.24841846525669098, 0.2508545219898224, 0.2513842284679413 ], "sources": 4 }, { "entropy_mean": 1.38628089427948, "mean_weights": [ 0.248581200838089, 0.24957583844661713, 0.24980047345161438, 0.2520424723625183 ], "sources": 4 }, { "entropy_mean": 1.386273741722107, "mean_weights": [ 0.2486811876296997, 0.2481946051120758, 0.2515612244606018, 0.25156301259994507 ], "sources": 4 }, { "entropy_mean": 1.3862909078598022, "mean_weights": [ 0.24919962882995605, 0.24966104328632355, 0.25073325634002686, 0.25040602684020996 ], "sources": 4 }, { "entropy_mean": 1.6094350814819336, "mean_weights": [ 0.20043906569480896, 0.20025089383125305, 0.1992931365966797, 0.19971072673797607, 0.20030616223812103 ], "sources": 5 }, { "entropy_mean": 1.6094260215759277, "mean_weights": [ 0.1990065574645996, 0.19869452714920044, 0.20100653171539307, 0.20066148042678833, 0.20063093304634094 ], "sources": 5 }, { "entropy_mean": 1.609431266784668, "mean_weights": [ 0.19923549890518188, 0.20021051168441772, 0.20099328458309174, 0.2002217173576355, 0.19933897256851196 ], "sources": 5 }, { "entropy_mean": 1.609424114227295, "mean_weights": [ 0.19919060170650482, 0.19957327842712402, 0.1993737816810608, 0.1998673677444458, 0.20199497044086456 ], "sources": 5 }, { "entropy_mean": 1.7917512655258179, "mean_weights": [ 0.16589663922786713, 0.16579726338386536, 0.16652584075927734, 0.1674002707004547, 0.1673392355442047, 0.16704075038433075 ], "sources": 6 }, { "entropy_mean": 1.7917526960372925, "mean_weights": [ 0.1660098284482956, 0.1660982221364975, 0.16643531620502472, 0.16734513640403748, 0.16757234930992126, 0.16653914749622345 ], "sources": 6 }, { "entropy_mean": 1.7917439937591553, "mean_weights": [ 0.1664077788591385, 0.1667102724313736, 0.16575725376605988, 0.16558866202831268, 0.16724394261837006, 0.1682921051979065 ], "sources": 6 }, { "entropy_mean": 1.7917406558990479, "mean_weights": [ 0.1662176251411438, 0.16636696457862854, 0.165130153298378, 0.16684658825397491, 0.16694699227809906, 0.1684916913509369 ], "sources": 6 }, { "entropy_mean": 1.9459023475646973, "mean_weights": [ 0.14233610033988953, 0.1422998011112213, 0.14257268607616425, 0.1434200406074524, 0.14240685105323792, 0.14367292821407318, 0.14329159259796143 ], "sources": 7 }, { "entropy_mean": 1.9458985328674316, "mean_weights": [ 0.1425248682498932, 0.14280283451080322, 0.1430899053812027, 0.14388513565063477, 0.14334818720817566, 0.14153052866458893, 0.14281854033470154 ], "sources": 7 }, { "entropy_mean": 1.94590425491333, "mean_weights": [ 0.14222130179405212, 0.14271843433380127, 0.14239171147346497, 0.1434115171432495, 0.1425933688879013, 0.14311325550079346, 0.14355038106441498 ], "sources": 7 }, { "entropy_mean": 1.94590425491333, "mean_weights": [ 0.14283692836761475, 0.14230498671531677, 0.14244258403778076, 0.1435498595237732, 0.14231878519058228, 0.14314766228199005, 0.143399178981781 ], "sources": 7 }, { "entropy_mean": 2.079432964324951, "mean_weights": [ 0.12469614297151566, 0.12493360042572021, 0.12421932816505432, 0.12482892721891403, 0.12584659457206726, 0.12576377391815186, 0.12501835823059082, 0.12469328939914703 ], "sources": 8 }, { "entropy_mean": 2.0794296264648438, "mean_weights": [ 0.12482310831546783, 0.12494714558124542, 0.12461379170417786, 0.12449270486831665, 0.12426981329917908, 0.12538880109786987, 0.12516307830810547, 0.12630154192447662 ], "sources": 8 }, { "entropy_mean": 2.0794286727905273, "mean_weights": [ 0.12459714710712433, 0.12472857534885406, 0.12479612231254578, 0.12524385750293732, 0.12565991282463074, 0.1262103021144867, 0.12457992881536484, 0.12418416142463684 ], "sources": 8 }, { "entropy_mean": 2.0794296264648438, "mean_weights": [ 0.12493155896663666, 0.12483802437782288, 0.12460187077522278, 0.12459874153137207, 0.12501779198646545, 0.12420021742582321, 0.12559065222740173, 0.12622115015983582 ], "sources": 8 }, { "entropy_mean": 2.197211265563965, "mean_weights": [ 0.11090581119060516, 0.11095649003982544, 0.11032625287771225, 0.11037011444568634, 0.11067523062229156, 0.11147095263004303, 0.11175340414047241, 0.1119384914636612, 0.1116032600402832 ], "sources": 9 }, { "entropy_mean": 2.1972126960754395, "mean_weights": [ 0.11083438992500305, 0.11066233366727829, 0.1106841117143631, 0.11084605753421783, 0.11061468720436096, 0.11108818650245667, 0.1115877777338028, 0.11133615672588348, 0.11234630644321442 ], "sources": 9 }, { "entropy_mean": 2.1972174644470215, "mean_weights": [ 0.11089816689491272, 0.1108870804309845, 0.11106622219085693, 0.11082258075475693, 0.11085513234138489, 0.1119755208492279, 0.11102905869483948, 0.11071982979774475, 0.11174643039703369 ], "sources": 9 } ], "layer_input_rms_by_sublayer": [ 0.0282417144626379, 0.01428750716149807, 0.014416527934372425, 0.014703329652547836, 0.014682094566524029, 0.01008235290646553, 0.010167798958718777, 0.010738944634795189, 0.01082152035087347, 0.00878758542239666, 0.008862113580107689, 0.009672283194959164, 0.00976790115237236, 0.008560523390769958, 0.008684799075126648, 0.00977813359349966, 0.009954164735972881, 0.008888072334229946, 0.00902888085693121, 0.01007006410509348, 0.010184217244386673, 0.009424775838851929, 0.009566618129611015, 0.010299356654286385, 0.010436818934977055, 0.009499363601207733, 0.009606627747416496, 0.010012919083237648, 0.010090752504765987, 0.00939567293971777, 0.009533073753118515, 0.009944267570972443 ], "loss_nats": 4.683111190795898, "loss_scale": 1.0, "output_weights": { "entropy_mean": 2.1972169876098633, "mean_weights": [ 0.11062228679656982, 0.11076737940311432, 0.1108340471982956, 0.11088211834430695, 0.11121642589569092, 0.11167430132627487, 0.111720010638237, 0.11070731282234192, 0.1115761250257492 ], "sources": 9 }, "step": 20, "stream_state_rms_by_sublayer": [ 0.00344275776296854, 0.005180664826184511, 0.006782237906008959, 0.007948367856442928, 0.0057159834541380405, 0.0071101076900959015, 0.010661935433745384, 0.011628935113549232, 0.008758665062487125, 0.009834222495555878, 0.014840804971754551, 0.015592618845403194, 0.010700107552111149, 0.011937727220356464, 0.019919002428650856, 0.02099907398223877, 0.012833327986299992, 0.014203419908881187, 0.022593334317207336, 0.023388667032122612, 0.012848040089011192, 0.014353072270751, 0.0218671802431345, 0.022941188886761665, 0.012859554030001163, 0.013975659385323524, 0.020060954615473747, 0.02055426314473152, 0.01440997514873743, 0.015514259226620197, 0.021293850615620613, 0.022412775084376335 ] } ], "environment": { "autocast": "cuda-bfloat16-forward-fp32-cross-entropy", "compile": false, "compute_capability": [ 12, 0 ], "cublas_workspace_config": ":4096:8", "cuda": "12.8", "deterministic_algorithms": true, "gpu": "NVIDIA GeForce RTX 5090", "python": "3.10.14", "torch": "2.11.0+cu128" }, "evaluations": [ { "bits_per_byte": 8.073347956844582, "cross_entropy_nats": 5.596018373966217, "step": 0 }, { "bits_per_byte": 6.7577811156122, "cross_entropy_nats": 4.684136927127838, "step": 20 } ], "gradient_gate": { "first_to_last_ratio_abs_delta": 0.0, "max_abs_scale_ratio_error": 0.0, "normalized_spectrum_max_abs_delta": 0.0, "passed": true, "per_block_scale_ratios": [ 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0 ], "population_cv_abs_delta": 0.0, "thresholds": { "scale_ratio_abs": 1e-05, "shape_abs": 1e-06 } }, "hashes": { "final_mixer_parameters": "5dd01b8270ff98adc6b1340e38ee82b56d324c449cbdbb97581afacdec1223dc", "final_model_state": "cf7d109dd2a8a9a3c903435afe8c733b4191e11d3150f7eee30962a8de6e0a3d", "final_optimizer_state": "96110c28f2690589414ee6ed445333520dd0bc9ac87eeb7d4f316b17494d8f30", "final_public_parameters": "b20e9b701d001f45efdba361481e6a18c3eb9f6fec33fa6434bd9773667dd881", "initial_mixer_parameters": "64fe0b90f5ea84ed88b6d93838558443c700c2b89ade2455ed526bd8db5d506f", "initial_public_parameter_elements": 9541824, "initial_public_parameter_structure": "e732db766f25e01f6ce1772cc182ced9de2c56c4a2384130117242b9444f0abe", "initial_public_parameter_tensors": 115, "initial_public_parameters": "af2724a1c34bcfd61d8a8bef402246430898c815e56b6e5c5257949a4eb0e7b1" }, "manifest": { "diagnostic_tensor_sha256": "21117e31db302b10d67b63f035665dc8f220b879d216ccd12b7d2ba86e7b1716", "file_sha256": "080afb17d1e036c0bba0a799fdb8b98ee4ad652bd42dd1b3b67110dd2ede6371", "formal_schedule_sha256": "5041e09b167f229248d2462324e8c254b8f5938975f135dcd8192b00a54a4f4e", "input_gate_tensor_hashes": { "0": "65136111a29a042e61a7909132560d95cd4bcf0f9b52f64d0fb2e57773856434", "1": "d995676b4e7dec8f661cd8c2345fe7fc7a513c17f528c02fc946a441a6995a94", "7999": "2345e7ac3decca2bdaebf13094fdc92fcabef42e3e461a2fefd6e8e7e76baccc" }, "path": "experiments/k3/attnres_gradient/manifest.json", "validation_tensor_sha256": "f459316f13078a163b47c133511bb7181e05170ab89516e196490113893ce338" }, "model": { "attnres_aggregation_groups": 8, "context": 256, "d_ff": 768, "d_head": 32, "d_model": 192, "heads": 6, "layers": 16, "parameters": { "core": 9541824, "embedding": 98304, "mixer": 12672, "total": 9554496 }, "sublayers": 32, "sublayers_per_attnres_group": 4, "transformer_blocks_per_attnres_group": 2, "vocabulary": 256 }, "optimizer": { "betas": [ 0.9, 0.95 ], "epsilon": 1e-08, "grad_clip": 1.0, "min_lr": 3e-05, "name": "AdamW", "peak_lr": 0.0003, "warmup_steps": 400, "weight_decay_ndim_ge_2": 0.1 }, "protocol_id": "llm-atlas-k3-attnres-gradient-scale-v1", "run_kind": "smoke", "schema_version": 1, "seed": 2026073001, "steps": 20, "target_bytes_seen": 163840, "timing": { "mean_ms": null, "measured_steps": 0, "median_ms": null, "p95_ms": null, "peak_allocated_bytes": 3462317568, "peak_reserved_bytes": 6889144320, "warmup_steps_excluded": 20 }, "training_history": [ { "bits_per_byte": 8.084436624250287, "learning_rate": 7.499999999999999e-07, "loss_nats": 5.603704452514648, "step": 1, "unclipped_grad_norm": 18.78145980834961 }, { "bits_per_byte": 7.484355446790379, "learning_rate": 7.499999999999999e-06, "loss_nats": 5.187759876251221, "step": 10, "unclipped_grad_norm": 12.867269515991211 }, { "bits_per_byte": 6.795359238834398, "learning_rate": 1.4999999999999999e-05, "loss_nats": 4.710184097290039, "step": 20, "unclipped_grad_norm": 4.549929141998291 } ] }