diff --git a/experiments/k3/attnres_spike/manifest.json b/experiments/k3/attnres_spike/manifest.json new file mode 100644 index 0000000..9df0064 --- /dev/null +++ b/experiments/k3/attnres_spike/manifest.json @@ -0,0 +1,118 @@ +{ + "schema_version": 1, + "protocol_id": "llm-atlas-k3-attnres-spike-path-v1", + "parent_protocol_id": "llm-atlas-k3-attnres-gradient-scale-v1", + "study_identity": "targeted follow-up informed by Round 05; not blind discovery", + "architecture": "block", + "depth": 32, + "aggregation_groups": 8, + "formal_seeds": [ + 2026073001, + 2026073002, + 2026073003 + ], + "replay": { + "architecture": "block", + "depth": 32, + "seed": 2026073001 + }, + "training": { + "steps": 8000, + "batch_size": 32, + "context": 256, + "target_bytes_per_cell": 65536000, + "diagnostic_steps": [ + 0, + 100, + 500, + 2000, + 4000, + 8000 + ] + }, + "positions": [ + "pre_attention_input", + "attention_branch_output", + "post_attention_state", + "pre_mlp_input", + "mlp_branch_output", + "post_mlp_state" + ], + "reductions": { + "confirmatory": [ + "element_rms", + "token_rms_mean", + "token_rms_median", + "token_rms_p95" + ], + "exploratory": [ + "batch_mean_rms", + "token_mean_rms" + ], + "algebraic_control": [ + "global_l2" + ], + "quantile_definition": "Hyndman-Fan Type 7 linear interpolation on explicitly sorted values" + }, + "interventions": { + "modes": [ + "learned", + "detached_learned", + "uniform_value_backward" + ], + "steps": [ + 0, + 8000 + ], + "scope": "all 64 depth mixers plus the output mixer", + "training_uses_custom_autograd": false, + "confirmatory_reduction": "element_rms" + }, + "spike_layers_one_based": [ + 21, + 22, + 23, + 24, + 25 + ], + "thresholds": { + "spike_contrast": 1.5, + "top_five_min_overlap": 3, + "spearman_minimum": 0.8, + "material_relative_drop": 0.2, + "spectrum_absolute_tolerance": 1e-06, + "raw_relative_tolerance": 1e-06, + "positive_denominator_epsilon": 1e-30 + }, + "parent_artifacts": { + "manifest_path": "experiments/k3/attnres_gradient/manifest.json", + "manifest_sha256": "080afb17d1e036c0bba0a799fdb8b98ee4ad652bd42dd1b3b67110dd2ede6371", + "runner_path": "experiments/k3/attnres_gradient/train.py", + "runner_sha256": "04ae69e10c58972c9193c2d31c7e09d924a0d0e834107afa4ba128c64ac5800f", + "protocol_path": "research/K3_ATTNRES_GRADIENT_SCALE_PROTOCOL.md", + "protocol_sha256": "f772629b3b82975b6756721c3a3bb57cc4171dfa26ba5b1e8043dc91c9dcce22", + "formal_schedule_sha256": "5041e09b167f229248d2462324e8c254b8f5938975f135dcd8192b00a54a4f4e", + "validation_tensor_sha256": "f459316f13078a163b47c133511bb7181e05170ab89516e196490113893ce338", + "diagnostic_tensor_sha256": "21117e31db302b10d67b63f035665dc8f220b879d216ccd12b7d2ba86e7b1716" + }, + "round05_expected": { + "2026073001": { + "raw_file_sha256": "29e1d638b481619c7b32de402122523db8b17cc1fc67a8881fa1ba132a1d5c38", + "canonical_sha256": "c7554d9beea6fe9e63617aafd88403f7fc93b4b305a0c6d603a98a8572e5b5f1", + "final_model_state": "3f0b97ece3a15571ba3d656f589f512ca0bb9e20083c9f58a42ccaee14892f59", + "final_optimizer_state": "ed03e6fbd4a12d8b063dcb22e0437754285f54d585374cd52fbd534f05d24637" + }, + "2026073002": { + "raw_file_sha256": "21199deb2199395061e51220fd8c7afd04a1135a6381e406da9b5795e3ad5032", + "canonical_sha256": "b4262e697e2269457fdebf31f75008383c6e8024ee1bbf96e5962fb2ba1dec18", + "final_model_state": "bd2556388aeaa211b798c283c7cbd8ccd29edf166a2922fa13d172e8dfdc38d1", + "final_optimizer_state": "0b101eab3bc7d8d654be2ea335c86fc25563ce19912d721844ee4e639c569e77" + }, + "2026073003": { + "raw_file_sha256": "c0d7f1bcfa9fa7f8f3134ca4571bdf23a951182d03b7de9611a6b4b89667d4ac", + "canonical_sha256": "513b85666fcdbf75424596e968cd73733314ec188345e453998de08b18dcef65", + "final_model_state": "638568aede21890773b6932a19ec4e112f5ac0a4770ba3402fcd82980a9ecf76", + "final_optimizer_state": "83947fd743ec8e3e31ca7788fd201981846f0f1c7e9e38c88afcef95cfc6ec4e" + } + } +} diff --git a/research/K3_ATTNRES_SPIKE_PROTOCOL.md b/research/K3_ATTNRES_SPIKE_PROTOCOL.md new file mode 100644 index 0000000..81018cb --- /dev/null +++ b/research/K3_ATTNRES_SPIKE_PROTOCOL.md @@ -0,0 +1,406 @@ +# K3 Attention Residuals 局部梯度尖峰与归约敏感性协议 + +协议 ID:`llm-atlas-k3-attnres-spike-path-v1` +冻结日期:2026-07-30 +协议状态:**结果前预注册** +父协议:`llm-atlas-k3-attnres-gradient-scale-v1` + +## 0. 研究身份 + +本轮是 Round 05 的**定向机制追踪**,不是盲发现: + +- 已知 depth-32 / Block 的三 seed 平均 layer 21–25 normalized post-MLP + activation-gradient RMS 较高; +- 已知 observationally,MLP mixer latest-source mass 与该梯度谱相关; +- 未知尖峰最早在哪个 block position 出现; +- 未知 mixer 的 softmax/key derivative path 与 learned value coefficients 分别贡献多少; +- 未知更换 raw gradient tensor 的公开 reduction 后,layer 21–25 是否仍构成稳定局部峰。 + +前置已知结果和相关分析固定在 +`research/K3_ATTNRES_SPIKE_SCOPING.md`。任何 Round 06 输出不得被倒写成“事前未知”。 + +## 1. 允许回答的问题 + +1. 在同一个缩小 Block AttnRes 模型里,layer 21–25 的相对高值在哪些 + attention / MLP 位置已经可见? +2. 在前向完全相同的条件下,对**全部 mixer** 切断 weight / key 导数路径是否降低尖峰? +3. 在前向完全相同的条件下,把**全部 mixer** 的 source value 反向系数改为均匀权重, + 是否降低尖峰? +4. element RMS、token-RMS mean / median / p95 是否给出一致的 spike 结论? +5. Round 06 的新增观测是否保持 Round 05 训练状态、BPC 和 post-MLP global RMS exact? + +## 2. 明确不回答的问题 + +- 论文 Figure 5(c) 的未公开 telemetry 定义; +- Kimi K3 2.8T checkpoint 的真实梯度; +- learned mixer 对大模型最终能力的因果收益; +- 改写 backward rule 后重新训练会发生什么; +- 哪个 source 具有可命名的语义; +- 三个 seed 之外的总体显著性或置信区间; +- 同 FLOPs、wall time 或参数量公平性。 + +## 3. 冻结训练合同 + +Round 06 不创建新训练任务分布,完整复用 Round 05: + +| 字段 | 固定值 | +|---|---| +| architecture | Block AttnRes | +| Transformer depth | 32 | +| aggregation groups | 8 | +| blocks / group | 4 | +| width / heads / FFN | 192 / 6 / 768 | +| context / vocabulary | 256 / byte-256 | +| formal seeds | 2026073001 / 2026073002 / 2026073003 | +| steps / batch | 8,000 / 32 | +| target bytes / formal cell | 65,536,000 | +| optimizer | AdamW | +| peak / min LR | 3e-4 / 3e-5 | +| warmup | 400 | +| weight decay | 0.1 for ndim ≥ 2 | +| betas / epsilon | 0.9, 0.95 / 1e-8 | +| clip | global norm 1.0 | +| forward | CUDA BF16 autocast | +| residual accumulation | explicit FP32 | +| diagnostic CE | fixed 16 × 256 token-mean FP32 CE | +| diagnostic steps | 0 / 100 / 500 / 2,000 / 4,000 / 8,000 | + +数据 bytes、training schedule、validation tensor 和 diagnostic tensor hashes 必须与父协议 +manifest exact。训练窗口仍由父协议 ID 派生;Round 06 ID 只标识新增 telemetry,不能改变 +任一 optimizer input。 + +## 4. 正式网格与 replay + +正式运行: + +```text +depth-32 / block / seed-2026073001 +depth-32 / block / seed-2026073002 +depth-32 / block / seed-2026073003 +``` + +另从初始化完整重跑: + +```text +replay / depth-32 / block / seed-2026073001 +``` + +正式三格共处理 196,608,000 target bytes;含 replay 共 262,144,000 bytes。 + +每格都必须在全新 Python 进程中运行。可并行两个进程,但不能共享 model、optimizer、RNG +或 CUDA graph。性能计时不进入数值复现合同。 + +## 5. 六个 activation 位置 + +对每个 Transformer block `l` 捕获: + +1. `pre_attention_input`:attention RMSNorm 的输入;Block 中是 attention mixer 输出; +2. `attention_branch_output`:attention projection 输出; +3. `post_attention_state`:attention branch 加入后的 FP32 partial state; +4. `pre_mlp_input`:MLP RMSNorm 的输入;Block 中是独立 MLP mixer 输出; +5. `mlp_branch_output`:MLP down projection 输出; +6. `post_mlp_state`:MLP branch 加入后的 FP32 partial state,即 Round 05 主对象。 + +每个位置必须有 32 个不同计算节点,shape 为 `[16,256,192]`,梯度全部 present / finite。 +跨位置允许计算图语义上的共享来源,但同一位置的 32 个条目不得意外复用同一 storage。 + +捕获在一次 learned-mode diagnostic forward / backward 中完成;不能把六个位置拆成六次 +不同 loss 的 backward: + +- `attention_branch_output` / `mlp_branch_output` 在原 module 输出、转成 FP32 residual + 之前捕获,保留其实际 autocast dtype; +- `post_attention_state` / `post_mlp_state` 在显式 FP32 residual accumulation 后捕获; +- 所有 `retain_grad()` 只允许出现在 diagnostic `capture=True` 路径; +- 8,000 个 optimizer training steps 必须走父 runner 的原始 `DepthMixer.forward`, + `capture=False`,不得进入 intervention custom autograd; +- 每次 diagnostic / intervention 前后都 `zero_grad(set_to_none=True)`; +- diagnostic 前后 optimizer-state tensor hash 必须 exact。 + +### 5.1 尖峰可见位置 + +固定目标集合: + +```text +S = layers 21, 22, 23, 24, 25 +R = other 27 layers +``` + +对每个位置和 seed: + +```text +spike_contrast = mean(metric[S]) / mean(metric[R]) +peak_normalized = max(metric) / mean(metric) +``` + +主判定**只使用 step 8,000**。若某个位置 `spike_contrast ≥ 1.5` 在 3 / 3 seed +成立,则称“该位置已可见局部峰”。六个位置按上述 forward 顺序报告;第一个满足者只标为 +**earliest tensor where the pattern is observed**,禁止写成 origin、injection point 或 +“在该算子生成”。其他 diagnostic steps 只展示轨迹,不参与位置判定。若没有位置 3 / 3 +达标,结论为 position-mixed。 + +## 6. 同一 raw gradient tensor 的 reductions + +令某层某位置梯度为 `g ∈ R[B,T,D]`,其中 `B=16,T=256,D=192`。 + +### 6.1 主 sensitivity family + +1. `element_rms = sqrt(mean_btd(g²))` +2. `token_rms_mean = mean_bt(sqrt(mean_d(g²)))` +3. `token_rms_median = median_bt(sqrt(mean_d(g²)))` +4. `token_rms_p95 = p95_bt(sqrt(mean_d(g²)))` + +这四项都测每 Token 梯度长度的分布,只改变平方根与 Token reduction 的次序/统计量。 + +median / p95 统一使用排序后的 **Hyndman–Fan Type 7 linear interpolation**: + +```text +h = (N - 1) × p +j = floor(h) +q_p = x_sorted[j] + (h - j) × (x_sorted[j + 1] - x_sorted[j]) +``` + +`p=.5/.95`,零基下标;若 `h` 为整数则直接取 `x_sorted[h]`。实现不得依赖 numpy / +torch 版本相关的默认 quantile 方法。 + +### 6.2 cancellation-sensitive diagnostics + +5. `batch_mean_rms = sqrt(mean_td((mean_b g)²))` +6. `token_mean_rms = sqrt(mean_bd((mean_t g)²))` + +它们允许正负梯度先抵消,测的是更相干的方向信号,只作探索性诊断,不进入主 robustness +判定。 + +### 6.3 代数控制 + +7. `global_l2 = sqrt(sum_btd(g²))` + +固定 shape 下它应满足: + +```text +global_l2 = element_rms × sqrt(B×T×D) +``` + +逐层 normalized spectrum、CV、spike contrast 应与 element RMS 在 `1e-6` 内相同。 + +### 6.4 reduction robustness 闸门 + +对最终 `post_mlp_state`,每个主 family reduction、每个 seed 必须同时满足: + +1. `spike_contrast ≥ 1.5`; +2. 该 reduction 的 top-5 layers 与固定集合 `S` 至少重合 3 层; +3. 与 element RMS 的 32-layer Spearman `ρ ≥ 0.8`。 + +四种 reductions、三个 seed 全部满足才记为 +`robust within the preregistered reduction family`。任何一格失败即为 `mixed`; +全数不满足才记为 `not robust at this threshold`。不添加事后替代阈值。 + +top-5 固定按 `metric descending, layer index ascending` 排序;Spearman 对并列值使用 +average ranks。cancellation-sensitive diagnostics 不得进入本节判定。 + +## 7. mixer backward-path interventions + +三种模式使用完全相同的 learned forward weights 与 model state。intervention 的作用域固定 +为**全部 64 个 depth mixers 加最终 output mixer**;不允许结果后只改 layer 21–25 邻域。 +因此本节回答的是“全局改写 mixer backward rule 后,固定尖峰指标是否变化”,不是把某个 +局部 mixer 宣布为唯一原因。 + +### 7.1 `learned` + +原始 mixer: + +```text +w = softmax(qᵀ RMSNorm(sources)) +y = Σ w_i source_i +``` + +反向同时经过 value coefficients 和 softmax / query / key 路径。 + +### 7.2 `detached_learned` + +前向仍用同一 `w`,但 `w` 在反向中 detach: + +```text +y = Σ stopgrad(w_i) source_i +``` + +它保留 learned value coefficients,切断 softmax / query / key 对 source gradient 的路径。 + +### 7.3 `uniform_value_backward` + +使用自定义 autograd: + +- 先调用父 runner 原始 `DepthMixer.forward` 得到 `y_parent`; +- custom Function 的 forward 直接返回已经计算出的 `y_parent`,不重算或做 + `y + z - z` 式浮点抵消; +- backward 对每个 source 返回 `grad_y / N`; +- 不向 learned weights / query / key 回传。 + +因此 forward logits、loss、所有 activation 值必须与 `learned` **逐元素 exact**,但 source +value 的 backward coefficient 变为均匀。 + +`detached_learned` 同样以 custom Function 原样返回 `y_parent`,backward 才按 learned +`w_i` 把 `grad_y` 分配给 source。两种 custom Function 都只允许在 step 0 / 8,000 +diagnostic 使用。 + +这些是**全局 backward-rule sensitivity diagnostics**,不是可训练模型变体,也不声称是 +合理部署方案。 + +## 8. intervention 判定 + +只在 step 0 和 8,000 运行三模式。 + +### 8.1 forward identity gate + +同 seed / step 三模式必须满足: + +- logits tensor SHA-256 exact; +- loss FP32 value exact; +- 六位置 activation tensor hashes exact; +- mixer forward mean / quantile summaries exact。 + +任一失败,正式结果无效。 + +learned mode 还必须证明:训练和普通 diagnostic 的每个 mixer output 使用父 runner 数值 +路径;custom Function 不得被训练 step 调用。 + +### 8.2 初始化负控制 + +step 0 的 mixer query 全为零,learned weights 是均匀分布。三模式的六位置 +`element_rms`: + +- 32 个 raw values 必须 finite 且严格大于 0; +- raw spectra 的逐层相对差必须 ≤ `1e-6`; +- normalized spectra 最大绝对差必须 ≤ `1e-6`。 + +失败则说明 intervention 实现没有隔离预期路径。 + +### 8.3 softmax / key derivative path + +最终 `post_mlp_state`,且 reduction 固定为 `element_rms`: + +```text +relative_drop_contrast = + (contrast_learned - contrast_detached) / contrast_learned + +relative_drop_peak = + (peak_learned - peak_detached) / peak_learned +``` + +`contrast_learned` 与 `peak_learned` 必须 finite 且 `>1e-30`,否则本格 invalid 并停止 +聚合。若两项都 `≥20%` 且 3 / 3 seed 同向,记为: + +> 全局切断所有 mixer 的 softmax / query / key source-gradient path,在本阈值下对固定 +> 尖峰指标有 material sensitivity。 + +禁止缩写成“layer 21–25 由 softmax/key path 造成”。 + +### 8.4 learned value coefficients + +用 `detached_learned → uniform_value_backward` 的同一公式,分母也必须 finite 且 +`>1e-30`。若 contrast 和 peak 都下降 `≥20%` 且 3 / 3 seed 同向,记为: + +> 全部 mixer 的 value backward coefficients 从 learned `w` 改成 `1/N`,在本阈值下 +> 对固定尖峰指标有 material sensitivity。 + +`N` 随 source count 变化;均匀系数不等于各 source 的数值贡献均匀。禁止写成“learned +value weights 是尖峰唯一原因”。 + +未过阈值只表示本协议没有达到“material”规则,不证明路径贡献为零。seed 或两个指标方向 +分裂时统一记为 `mixed`。 + +## 9. mixer 分布摘要 + +每个子层 mixer 保存: + +- source count; +- source labels; +- 每 source mean / p05 / median / p95; +- entropy mean; +- normalized entropy; +- max source mass; +- latest-source mass; +- group / layer / attention-or-MLP 身份。 + +相关分析报告 Pearson 与 Spearman,但永远标为 association;只有第 7–8 节的同前向 +backward intervention 可以支持路径贡献判断。 + +不保存可还原语料内容的逐 Token weight arrays。 + +## 10. 训练等价与复现闸门 + +每个 formal seed 必须与对应 Round 05 depth-32 / Block raw 结果满足: + +- final model-state hash exact; +- final optimizer-state hash exact; +- 六个 validation BPC exact; +- 六个 Round 05 post-MLP element-RMS arrays exact; +- training-history 冻结字段 exact。 + +此外,step-0 smoke 对六个位置逐一执行同一 loss 的 `×1 / ×2` backward: + +- 每层每 reduction 中线性尺度应乘 2 的指标,其比值在 `2±1e-5`; +- normalized spectrum、CV、spike contrast 的绝对差 ≤ `1e-6`; +- shape、dtype、gradient presence / finite 与同位置 storage uniqueness 全部过闸。 + +普通 diagnostic 与三种 intervention 前后 optimizer-state hash 必须 exact;训练 step +禁止 `capture=True`。 + +seed-1 replay 还必须与 Round 06 formal seed-1 的以下字段 exact: + +- 全部训练等价字段; +- 六位置、七 reductions、三 intervention 的全部数值; +- mixer quantile summaries; +- logits / activation tensor hashes; +- final model / optimizer hashes。 + +排除: + +- run kind / output path; +- wall time / step time; +- peak allocated / reserved memory; +- process / host 瞬时字段。 + +任一训练等价字段失败,不能聚合机制结果;replay 失败则公开失败并停止网站结论。 + +## 11. 正式工件 + +```text +experiments/k3/attnres_spike/ + README.md + manifest.json + scope.py + train.py + analyze.py + reproduction.json + results/raw/*.json + +research/ + K3_ATTNRES_SPIKE_SCOPING.md + K3_ATTNRES_SPIKE_PROTOCOL.md + K3_ATTNRES_SPIKE_AUDIT.md + +src/data/ + k3-attnres-spike.json + k3-attnres-spike-compact.json +``` + +网站必须把四种证据身份分开: + +- Round 05 已知峰; +- observational mixer association; +- same-forward backward intervention; +- reduction robustness / failure。 + +## 12. 停止规则 + +出现以下任一情况立即停止正式聚合: + +- 训练 hash 不匹配 Round 05; +- forward identity gate 失败; +- step-0 uniform negative control 失败; +- 六位置 loss×2 或 optimizer isolation gate 失败; +- activation gradient 缺失、非 finite 或 shape 错误; +- replay 数值不 exact; +- 正式 raw 文件缺失或 canonical hash 不闭合。 + +遇到反结果时不修改集合 `S`、20% 阈值、1.5 contrast、top-5 overlap 或 Spearman 阈值。 diff --git a/research/K3_ATTNRES_SPIKE_SCOPING.md b/research/K3_ATTNRES_SPIKE_SCOPING.md new file mode 100644 index 0000000..49f3f4b --- /dev/null +++ b/research/K3_ATTNRES_SPIKE_SCOPING.md @@ -0,0 +1,120 @@ +# K3 Attention Residuals 局部梯度尖峰:Round 06 前置定位 + +研究日期:2026-07-30 +阶段身份:**探索性 scoping,不是 Round 06 预注册结果** +上游协议:`llm-atlas-k3-attnres-gradient-scale-v1` + +## 1. 为什么先做 scoping + +Round 05 已经公开了一个方向分裂: + +- Block AttnRes 在 depth 16 / 32、三个 seed 的 6 / 6 配对中都改善首/末四分位失衡; +- 但它也在 6 / 6 配对中提高全层 activation-gradient CV; +- depth 32 的三 seed 平均峰值集中在 layer 21–25。 + +这意味着下一步不该再问一个笼统的“梯度是否均匀”,而应该问: + +1. 尖峰在 attention / MLP block 的哪个位置已经出现? +2. 它是否与 learned mixer 的 source 权重有关? +3. 相关来自 mixer softmax / key 的导数路径,还是来自 learned value coefficients? +4. 换一种公开、合理的 gradient reduction 后,尖峰是否还存在? + +Round 05 已保存每个诊断点的逐层 post-MLP gradient RMS 和 64 个子层 mixer +分布。这里先用那些**已经见过的数据**定位假设,再冻结新协议。Round 06 的因果式 +backward-path intervention 不能被描述成盲验证。 + +## 2. 对齐规则 + +depth 32 的 Block AttnRes 使用 8 个 aggregation groups,每组 4 个 Transformer +blocks、8 个 attention / MLP residual sublayers。 + +对 1-based Transformer layer `l`: + +```text +attention mixer index = 2 × (l - 1) +MLP mixer index = 2 × (l - 1) + 1 +group = floor((l - 1) / 4) + 1 +offset in group = ((l - 1) mod 4) + 1 +``` + +`latest source mass` 定义为对应 mixer `mean_weights` 的最后一个元素。它在不同 +group offset 的语义并不完全相同: + +- group 首层 attention mixer 没有 current partial,最后源是最近完成的 group; +- 其他 attention mixer 的最后源是 current partial; +- 每层 MLP mixer 的最后源都是刚加入 attention branch 后的 current partial。 + +因此不能把所有 `last` 自动命名为同一种“最近层贡献”。 + +## 3. layer 19–28 的三 seed 均值 + +下表的 gradient 是每个 seed 先按该 seed 的 32 层均值归一化,再跨 seed 平均。 +权重和 normalized entropy 也跨三个 seed 平均。 + +| layer | group / offset | gradient / layer mean | attn last | MLP last | attn H/logN | MLP H/logN | +|---:|---:|---:|---:|---:|---:|---:| +| 19 | 5 / 3 | 1.063 | 0.349 | 0.452 | 0.929 | 0.842 | +| 20 | 5 / 4 | 1.014 | 0.316 | 0.419 | 0.946 | 0.863 | +| 21 | 6 / 1 | **3.042** | 0.250 | **0.513** | 0.969 | 0.763 | +| 22 | 6 / 2 | **2.405** | 0.436 | **0.563** | 0.851 | 0.721 | +| 23 | 6 / 3 | **1.910** | 0.400 | **0.551** | 0.879 | 0.737 | +| 24 | 6 / 4 | **1.548** | 0.384 | **0.515** | 0.889 | 0.771 | +| 25 | 7 / 1 | **1.769** | 0.317 | 0.358 | 0.934 | 0.766 | +| 26 | 7 / 2 | 1.369 | 0.343 | 0.420 | 0.876 | 0.764 | +| 27 | 7 / 3 | 0.986 | 0.325 | 0.402 | 0.891 | 0.775 | +| 28 | 7 / 4 | 0.802 | 0.296 | 0.304 | 0.902 | 0.822 | + +layer 21–24 正好是第 6 个 group,layer 25 是第 7 个 group 的首层。梯度峰并不 +只是一个 group boundary 单点;它在第 6 组内部递减,并在下一组首层出现较小的第二峰。 + +## 4. 相关线索 + +把三个 seed 的 32 层合成 96 个点,post-MLP normalized gradient 与 mixer 摘要的 +Pearson 相关为: + +| 变量 | 96 点 `r` | layer 19–28 的 30 点 `r` | +|---|---:|---:| +| attention latest-source mass | 0.066 | 0.158 | +| MLP latest-source mass | **0.651** | **0.690** | +| attention normalized entropy | −0.296 | −0.057 | +| MLP normalized entropy | −0.274 | **−0.636** | +| attention max source mass | −0.024 | 0.158 | +| MLP max source mass | 0.230 | **0.702** | + +这些数值只支持: + +> 尖峰层与更集中的 MLP mixer、尤其较大的 latest-partial mean weight 同时出现。 + +它们不支持: + +- “MLP latest weight 导致梯度尖峰”; +- “第 6 group 是唯一原因”; +- “降低 entropy 就一定增加梯度”; +- “K3 真实 checkpoint 也有相同模式”。 + +同一个 learned mixer 同时改变 forward activation、value-path gradient coefficient +和 softmax/key derivative path。单看 observational correlation 无法分解这三者。 + +## 5. Round 06 要冻结的可证伪问题 + +Round 06 将复跑与 Round 05 完全相同的 depth-32 / Block 三 seed 训练,并要求最终 +model / optimizer hash 与 Round 05 exact。新增诊断只在 optimizer step 之外执行。 + +它将同时测量: + +- pre-attention input; +- attention branch output; +- post-attention partial state; +- pre-MLP input; +- MLP branch output; +- post-MLP partial state; +- 同一 raw gradient tensor 的多种 reduction; +- learned backward; +- learned weights detached backward; +- learned forward + uniform value backward。 + +其中后两项保持**前向 logits、loss 与 activation 完全相同**,只改变反向路径。这样才能 +区分“相关”与“哪条 backward path 对尖峰有实质贡献”。 + +完整阈值、失败规则和复现合同见 +`research/K3_ATTNRES_SPIKE_PROTOCOL.md`。