{ "architecture": "baseline", "batch_size": 32, "canonical_sha256_without_self": "0ae9ad1697ca15ec4c84270ad82b9fa98056f14f20fc7372c4b9886c29845415", "depth": 16, "diagnostics": [ { "activation_grad_rms_by_block": [ 0.0002865509013645351, 0.00024785567075014114, 0.00021551651298068464, 0.00019683866412378848, 0.0001823508064262569, 0.0001679604029050097, 0.00015584587526973337, 0.00014737028686795384, 0.00014130656199995428, 0.00013612695329356939, 0.00013129493163432926, 0.00012807638267986476, 0.00012550120300147682, 0.00012324050476308912, 0.00012108208466088399, 0.00011888353037647903 ], "activation_grad_statistics": { "first_quartile_mean": 0.00023669043730478734, "first_to_last_ratio": 1.9372775995887173, "imbalance_abs_log_ratio": 0.6612836883477485, "last_quartile_mean": 0.00012217683070048224, "mean": 0.00016411257956860936, "normalized": [ 1.7460629899168627, 1.5102783187106137, 1.313223602646409, 1.1994124072706904, 1.1111324123086055, 1.0234462424910682, 0.9496278449793059, 0.8979828801383491, 0.8610343117596252, 0.8294729974472175, 0.8000296624393728, 0.7804178266926869, 0.7647262832098098, 0.7509509940495869, 0.7377989242455608, 0.7244023016942358 ], "population_cv": 0.29361872380036635 }, "activation_output_rms_by_block": [ 0.02866268903017044, 0.029099803417921066, 0.02953805774450302, 0.030088091269135475, 0.030644793063402176, 0.03131929785013199, 0.03213750571012497, 0.03286394104361534, 0.033869802951812744, 0.03499744459986687, 0.03588006645441055, 0.037365153431892395, 0.03789033368229866, 0.039445169270038605, 0.04050002992153168, 0.04121527820825577 ], "activation_output_statistics": { "first_quartile_mean": 0.0293471603654325, "first_to_last_ratio": 0.7380574840395957, "imbalance_abs_log_ratio": 0.3037335657624931, "last_quartile_mean": 0.03976270277053118, "mean": 0.034094841103069484, "normalized": [ 0.8406752488894867, 0.8534957922212247, 0.8663497699023965, 0.8824822259232266, 0.8988102619619861, 0.9185934539320196, 0.942591449919727, 0.9638977622528551, 0.9933996421752943, 1.0264733158329964, 1.0523605710888722, 1.0959180985456622, 1.1113216092650298, 1.1569248600043318, 1.1878638706395246, 1.2088420674453664 ], "population_cv": 0.11952987076282337 }, "bits_per_byte": 8.096463027059821, "branch_output_rms_by_sublayer": [ 0.0033055038657039404, 0.0037518907338380814, 0.0033751516602933407, 0.003911525942385197, 0.004229962360113859, 0.0038461871445178986, 0.004243654198944569, 0.0038366704247891903, 0.004372371360659599, 0.003851204412057996, 0.005078981164842844, 0.003818925702944398, 0.0054668826051056385, 0.0038749484810978174, 0.005823117680847645, 0.0037566579412668943, 0.0065947300754487514, 0.003876061411574483, 0.006763988174498081, 0.003795720636844635, 0.006863070651888847, 0.0038706199266016483, 0.007856340147554874, 0.003839300014078617, 0.00830968376249075, 0.003927029203623533, 0.009655521251261234, 0.0038513424806296825, 0.009004893712699413, 0.0039484030567109585, 0.008608299307525158, 0.003791053779423237 ], "capture": { "all_gradients_finite": true, "all_gradients_present": true, "count": 16, "dtypes": [ "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32" ], "position": "post-MLP Transformer-block output; Block AttnRes is captured before aggregation-partial reset", "shape": [ 16, 256, 192 ], "storage_unique": true }, "core_parameter_grad_rms_by_block": [ 0.00809059897248305, 0.007313812062277777, 0.007090710224412361, 0.0067033738367094624, 0.006535407867454384, 0.006785021795736036, 0.006295993569919972, 0.005509300326230647, 0.006231736048910802, 0.006047390980174007, 0.005629405359532731, 0.005784042737956458, 0.0059591736113607996, 0.0061983173180850636, 0.006571690039252182, 0.005060205437758869 ], "core_parameter_grad_statistics": { "first_quartile_mean": 0.007299623773970663, "first_to_last_ratio": 1.227374872012571, "imbalance_abs_log_ratio": 0.2048776382299483, "last_quartile_mean": 0.005947346601614228, "mean": 0.006362886261765913, "normalized": [ 1.271529717747562, 1.1494488132258316, 1.1143858200043408, 1.05351149791715, 1.0271137340180259, 1.066343404015675, 0.9894870520870547, 0.8658492545019393, 0.9793882512652816, 0.9504163254515969, 0.8847251275509024, 0.909028151691524, 0.9365519618304368, 0.9741361173356609, 1.0328158902888074, 0.7952688810682109 ], "population_cv": 0.11384977604868386 }, "depth_weights": [], "layer_input_rms_by_sublayer": [ 0.02823694422841072, 0.028404271230101585, 0.02866268903017044, 0.02886480651795864, 0.029099803417921066, 0.029365191236138344, 0.02953805774450302, 0.029881296679377556, 0.030088091269135475, 0.030441828072071075, 0.030644793063402176, 0.03106885403394699, 0.03131929785013199, 0.031892918050289154, 0.03213750571012497, 0.03266792371869087, 0.03286394104361534, 0.03364725783467293, 0.033869802951812744, 0.03485054895281792, 0.03499744459986687, 0.03562505170702934, 0.03588006645441055, 0.03719272464513779, 0.037365153431892395, 0.03768136352300644, 0.03789033368229866, 0.03927159309387207, 0.039445169270038605, 0.04041972756385803, 0.04050002992153168, 0.04110744968056679 ], "loss_nats": 5.6120405197143555, "loss_scale": 1.0, "output_weights": null, "step": 0, "stream_state_rms_by_sublayer": [ 0.028404271230101585, 0.02866268903017044, 0.02886480651795864, 0.029099803417921066, 0.029365191236138344, 0.02953805774450302, 0.029881296679377556, 0.030088091269135475, 0.030441828072071075, 0.030644793063402176, 0.03106885403394699, 0.03131929785013199, 0.031892918050289154, 0.03213750571012497, 0.03266792371869087, 0.03286394104361534, 0.03364725783467293, 0.033869802951812744, 0.03485054895281792, 0.03499744459986687, 0.03562505170702934, 0.03588006645441055, 0.03719272464513779, 0.037365153431892395, 0.03768136352300644, 0.03789033368229866, 0.03927159309387207, 0.039445169270038605, 0.04041972756385803, 0.04050002992153168, 0.04110744968056679, 0.04121527820825577 ] }, { "activation_grad_rms_by_block": [ 5.69471885683015e-05, 5.527245593839325e-05, 5.393734318204224e-05, 5.292302375892177e-05, 5.23102717124857e-05, 5.178990977583453e-05, 5.1564093155320734e-05, 5.129844430484809e-05, 5.121947833686136e-05, 5.116340616950765e-05, 5.109804988023825e-05, 5.104695082991384e-05, 5.09646451973822e-05, 5.100828275317326e-05, 5.106439857627265e-05, 5.11144389747642e-05 ], "activation_grad_statistics": { "first_quartile_mean": 5.477000286191469e-05, "first_to_last_ratio": 1.0731232762518041, "imbalance_abs_log_ratio": 0.07057334637995329, "last_quartile_mean": 5.103794137539808e-05, "mean": 5.2170148819641327e-05, "normalized": [ 1.0915665348238701, 1.0594651767139285, 1.033873669184083, 1.0144311441756324, 1.002685882559561, 0.9927115591500164, 0.988383094968431, 0.9832911246274795, 0.9817775010367221, 0.9807027069519371, 0.9794499543578176, 0.9784704852268963, 0.9768928467805085, 0.9777292936141551, 0.9788049244944386, 0.9797641013345227 ], "population_cv": 0.032815404485150704 }, "activation_output_rms_by_block": [ 0.028743742033839226, 0.02942933700978756, 0.030596865341067314, 0.03249936178326607, 0.03542664647102356, 0.03894684836268425, 0.04318666458129883, 0.04936420917510986, 0.05370106175541878, 0.06048284471035004, 0.0664696991443634, 0.07231792062520981, 0.07627613097429276, 0.07997097074985504, 0.08494995534420013, 0.08972473442554474 ], "activation_output_statistics": { "first_quartile_mean": 0.03031732654199004, "first_to_last_ratio": 0.3664591129538783, "imbalance_abs_log_ratio": 1.0038683247140607, "last_quartile_mean": 0.08273044787347317, "mean": 0.05450543703045696, "normalized": [ 0.5273555006590916, 0.5399339701348108, 0.5613543713807885, 0.5962590808162097, 0.6499653686150171, 0.7145497859400932, 0.7923368187501488, 0.9056749539963465, 0.9852422929002714, 1.1096662646068314, 1.21950584686113, 1.326801958945847, 1.399422427007981, 1.4672108895332452, 1.5585592919240543, 1.6461611779281335 ], "population_cv": 0.3808757530834043 }, "bits_per_byte": 6.7509657011324835, "branch_output_rms_by_sublayer": [ 0.003436450148001313, 0.003791899885982275, 0.003964710980653763, 0.003938332665711641, 0.0056790695525705814, 0.003919641952961683, 0.006960175931453705, 0.0038887569680809975, 0.008646421134471893, 0.003928068559616804, 0.009785857982933521, 0.0038193853106349707, 0.010625048540532589, 0.003996006678789854, 0.012892307713627815, 0.0039014811627566814, 0.012757784686982632, 0.003861474571749568, 0.014726920053362846, 0.003962590359151363, 0.012820222415030003, 0.004551138263195753, 0.01401793584227562, 0.004161422606557608, 0.01284075528383255, 0.004107494372874498, 0.012963366694748402, 0.0034911984112113714, 0.014392351731657982, 0.004168955609202385, 0.013408888131380081, 0.003909120801836252 ], "capture": { "all_gradients_finite": true, "all_gradients_present": true, "count": 16, "dtypes": [ "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32", "torch.float32" ], "position": "post-MLP Transformer-block output; Block AttnRes is captured before aggregation-partial reset", "shape": [ 16, 256, 192 ], "storage_unique": true }, "core_parameter_grad_rms_by_block": [ 0.0004507690637370199, 0.0004530669153600251, 0.0005174622333717066, 0.0005579942111201043, 0.0006494724560479704, 0.0007477190266084469, 0.0008433132451421535, 0.0008983201732199723, 0.001008715304343127, 0.0011143118429534733, 0.0010428854825857738, 0.0010916116075341876, 0.0011085001103672855, 0.0012404769719650786, 0.0013545740271648265, 0.0014243271893498689 ], "core_parameter_grad_statistics": { "first_quartile_mean": 0.000494823105897214, "first_to_last_ratio": 0.385986622193018, "imbalance_abs_log_ratio": 0.9519525676489338, "last_quartile_mean": 0.001281969574711765, "mean": 0.0009064699913044387, "normalized": [ 0.49727963204644987, 0.4998145771025995, 0.5708542349284638, 0.6155683215912455, 0.7164853357289404, 0.824869034585972, 0.930326710461313, 0.99100927977468, 1.112795033503044, 1.2292870736404011, 1.1504909071341995, 1.2042446170372658, 1.2228756836970622, 1.3684699812069823, 1.494339625314625, 1.571289952246756 ], "population_cv": 0.338812252534038 }, "depth_weights": [], "layer_input_rms_by_sublayer": [ 0.028241552412509918, 0.028533460572361946, 0.028743742033839226, 0.029236938804388046, 0.02942933700978756, 0.03035305254161358, 0.030596865341067314, 0.032195623964071274, 0.03249936178326607, 0.03506496921181679, 0.03542664647102356, 0.03852483630180359, 0.03894684836268425, 0.04251723736524582, 0.04318666458129883, 0.04861506074666977, 0.04936420917510986, 0.052915990352630615, 0.05370106175541878, 0.0598941408097744, 0.06048284471035004, 0.06538087129592896, 0.0664696991443634, 0.07142146676778793, 0.07231792062520981, 0.07521561533212662, 0.07627613097429276, 0.07935076206922531, 0.07997097074985504, 0.08368266373872757, 0.08494995534420013, 0.08860929310321808 ], "loss_nats": 4.679412841796875, "loss_scale": 1.0, "output_weights": null, "step": 20, "stream_state_rms_by_sublayer": [ 0.028533460572361946, 0.028743742033839226, 0.029236938804388046, 0.02942933700978756, 0.03035305254161358, 0.030596865341067314, 0.032195623964071274, 0.03249936178326607, 0.03506496921181679, 0.03542664647102356, 0.03852483630180359, 0.03894684836268425, 0.04251723736524582, 0.04318666458129883, 0.04861506074666977, 0.04936420917510986, 0.052915990352630615, 0.05370106175541878, 0.0598941408097744, 0.06048284471035004, 0.06538087129592896, 0.0664696991443634, 0.07142146676778793, 0.07231792062520981, 0.07521561533212662, 0.07627613097429276, 0.07935076206922531, 0.07997097074985504, 0.08368266373872757, 0.08494995534420013, 0.08860929310321808, 0.08972473442554474 ] } ], "environment": { "autocast": "cuda-bfloat16-forward-fp32-cross-entropy", "compile": false, "compute_capability": [ 12, 0 ], "cublas_workspace_config": ":4096:8", "cuda": "12.8", "deterministic_algorithms": true, "gpu": "NVIDIA GeForce RTX 5090", "python": "3.10.14", "torch": "2.11.0+cu128" }, "evaluations": [ { "bits_per_byte": 8.076786148009306, "cross_entropy_nats": 5.5984015464782715, "step": 0 }, { "bits_per_byte": 6.752401670275862, "cross_entropy_nats": 4.680408179759979, "step": 20 } ], "gradient_gate": { "first_to_last_ratio_abs_delta": 0.0, "max_abs_scale_ratio_error": 0.0, "normalized_spectrum_max_abs_delta": 0.0, "passed": true, "per_block_scale_ratios": [ 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0, 2.0 ], "population_cv_abs_delta": 0.0, "thresholds": { "scale_ratio_abs": 1e-05, "shape_abs": 1e-06 } }, "hashes": { "final_mixer_parameters": null, "final_model_state": "d53dab0fa7c76215ab91de676b2aef3f9ef14cb0cc1819b7c4a887915bed97c0", "final_optimizer_state": "a7bce44db1478ce53933758aa5033bbb1e0aa21296e9615c33146920f30a2057", "final_public_parameters": "d53dab0fa7c76215ab91de676b2aef3f9ef14cb0cc1819b7c4a887915bed97c0", "initial_mixer_parameters": null, "initial_public_parameter_elements": 9541824, "initial_public_parameter_structure": "e732db766f25e01f6ce1772cc182ced9de2c56c4a2384130117242b9444f0abe", "initial_public_parameter_tensors": 115, "initial_public_parameters": "af2724a1c34bcfd61d8a8bef402246430898c815e56b6e5c5257949a4eb0e7b1" }, "manifest": { "diagnostic_tensor_sha256": "21117e31db302b10d67b63f035665dc8f220b879d216ccd12b7d2ba86e7b1716", "file_sha256": "080afb17d1e036c0bba0a799fdb8b98ee4ad652bd42dd1b3b67110dd2ede6371", "formal_schedule_sha256": "5041e09b167f229248d2462324e8c254b8f5938975f135dcd8192b00a54a4f4e", "input_gate_tensor_hashes": { "0": "65136111a29a042e61a7909132560d95cd4bcf0f9b52f64d0fb2e57773856434", "1": "d995676b4e7dec8f661cd8c2345fe7fc7a513c17f528c02fc946a441a6995a94", "7999": "2345e7ac3decca2bdaebf13094fdc92fcabef42e3e461a2fefd6e8e7e76baccc" }, "path": "experiments/k3/attnres_gradient/manifest.json", "validation_tensor_sha256": "f459316f13078a163b47c133511bb7181e05170ab89516e196490113893ce338" }, "model": { "attnres_aggregation_groups": 8, "context": 256, "d_ff": 768, "d_head": 32, "d_model": 192, "heads": 6, "layers": 16, "parameters": { "core": 9541824, "embedding": 98304, "mixer": 0, "total": 9541824 }, "sublayers": 32, "sublayers_per_attnres_group": 4, "transformer_blocks_per_attnres_group": 2, "vocabulary": 256 }, "optimizer": { "betas": [ 0.9, 0.95 ], "epsilon": 1e-08, "grad_clip": 1.0, "min_lr": 3e-05, "name": "AdamW", "peak_lr": 0.0003, "warmup_steps": 400, "weight_decay_ndim_ge_2": 0.1 }, "protocol_id": "llm-atlas-k3-attnres-gradient-scale-v1", "run_kind": "smoke", "schema_version": 1, "seed": 2026073001, "steps": 20, "target_bytes_seen": 163840, "timing": { "mean_ms": null, "measured_steps": 0, "median_ms": null, "p95_ms": null, "peak_allocated_bytes": 1648265728, "peak_reserved_bytes": 3282042880, "warmup_steps_excluded": 20 }, "training_history": [ { "bits_per_byte": 8.088097790921855, "learning_rate": 7.499999999999999e-07, "loss_nats": 5.6062421798706055, "step": 1, "unclipped_grad_norm": 19.475919723510742 }, { "bits_per_byte": 7.474302716882146, "learning_rate": 7.499999999999999e-06, "loss_nats": 5.180791854858398, "step": 10, "unclipped_grad_norm": 12.938376426696777 }, { "bits_per_byte": 6.789615018295581, "learning_rate": 1.4999999999999999e-05, "loss_nats": 4.706202507019043, "step": 20, "unclipped_grad_norm": 4.482712745666504 } ] }