research: preregister AttnRes spike path study
This commit is contained in:
@@ -0,0 +1,118 @@
|
|||||||
|
{
|
||||||
|
"schema_version": 1,
|
||||||
|
"protocol_id": "llm-atlas-k3-attnres-spike-path-v1",
|
||||||
|
"parent_protocol_id": "llm-atlas-k3-attnres-gradient-scale-v1",
|
||||||
|
"study_identity": "targeted follow-up informed by Round 05; not blind discovery",
|
||||||
|
"architecture": "block",
|
||||||
|
"depth": 32,
|
||||||
|
"aggregation_groups": 8,
|
||||||
|
"formal_seeds": [
|
||||||
|
2026073001,
|
||||||
|
2026073002,
|
||||||
|
2026073003
|
||||||
|
],
|
||||||
|
"replay": {
|
||||||
|
"architecture": "block",
|
||||||
|
"depth": 32,
|
||||||
|
"seed": 2026073001
|
||||||
|
},
|
||||||
|
"training": {
|
||||||
|
"steps": 8000,
|
||||||
|
"batch_size": 32,
|
||||||
|
"context": 256,
|
||||||
|
"target_bytes_per_cell": 65536000,
|
||||||
|
"diagnostic_steps": [
|
||||||
|
0,
|
||||||
|
100,
|
||||||
|
500,
|
||||||
|
2000,
|
||||||
|
4000,
|
||||||
|
8000
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"positions": [
|
||||||
|
"pre_attention_input",
|
||||||
|
"attention_branch_output",
|
||||||
|
"post_attention_state",
|
||||||
|
"pre_mlp_input",
|
||||||
|
"mlp_branch_output",
|
||||||
|
"post_mlp_state"
|
||||||
|
],
|
||||||
|
"reductions": {
|
||||||
|
"confirmatory": [
|
||||||
|
"element_rms",
|
||||||
|
"token_rms_mean",
|
||||||
|
"token_rms_median",
|
||||||
|
"token_rms_p95"
|
||||||
|
],
|
||||||
|
"exploratory": [
|
||||||
|
"batch_mean_rms",
|
||||||
|
"token_mean_rms"
|
||||||
|
],
|
||||||
|
"algebraic_control": [
|
||||||
|
"global_l2"
|
||||||
|
],
|
||||||
|
"quantile_definition": "Hyndman-Fan Type 7 linear interpolation on explicitly sorted values"
|
||||||
|
},
|
||||||
|
"interventions": {
|
||||||
|
"modes": [
|
||||||
|
"learned",
|
||||||
|
"detached_learned",
|
||||||
|
"uniform_value_backward"
|
||||||
|
],
|
||||||
|
"steps": [
|
||||||
|
0,
|
||||||
|
8000
|
||||||
|
],
|
||||||
|
"scope": "all 64 depth mixers plus the output mixer",
|
||||||
|
"training_uses_custom_autograd": false,
|
||||||
|
"confirmatory_reduction": "element_rms"
|
||||||
|
},
|
||||||
|
"spike_layers_one_based": [
|
||||||
|
21,
|
||||||
|
22,
|
||||||
|
23,
|
||||||
|
24,
|
||||||
|
25
|
||||||
|
],
|
||||||
|
"thresholds": {
|
||||||
|
"spike_contrast": 1.5,
|
||||||
|
"top_five_min_overlap": 3,
|
||||||
|
"spearman_minimum": 0.8,
|
||||||
|
"material_relative_drop": 0.2,
|
||||||
|
"spectrum_absolute_tolerance": 1e-06,
|
||||||
|
"raw_relative_tolerance": 1e-06,
|
||||||
|
"positive_denominator_epsilon": 1e-30
|
||||||
|
},
|
||||||
|
"parent_artifacts": {
|
||||||
|
"manifest_path": "experiments/k3/attnres_gradient/manifest.json",
|
||||||
|
"manifest_sha256": "080afb17d1e036c0bba0a799fdb8b98ee4ad652bd42dd1b3b67110dd2ede6371",
|
||||||
|
"runner_path": "experiments/k3/attnres_gradient/train.py",
|
||||||
|
"runner_sha256": "04ae69e10c58972c9193c2d31c7e09d924a0d0e834107afa4ba128c64ac5800f",
|
||||||
|
"protocol_path": "research/K3_ATTNRES_GRADIENT_SCALE_PROTOCOL.md",
|
||||||
|
"protocol_sha256": "f772629b3b82975b6756721c3a3bb57cc4171dfa26ba5b1e8043dc91c9dcce22",
|
||||||
|
"formal_schedule_sha256": "5041e09b167f229248d2462324e8c254b8f5938975f135dcd8192b00a54a4f4e",
|
||||||
|
"validation_tensor_sha256": "f459316f13078a163b47c133511bb7181e05170ab89516e196490113893ce338",
|
||||||
|
"diagnostic_tensor_sha256": "21117e31db302b10d67b63f035665dc8f220b879d216ccd12b7d2ba86e7b1716"
|
||||||
|
},
|
||||||
|
"round05_expected": {
|
||||||
|
"2026073001": {
|
||||||
|
"raw_file_sha256": "29e1d638b481619c7b32de402122523db8b17cc1fc67a8881fa1ba132a1d5c38",
|
||||||
|
"canonical_sha256": "c7554d9beea6fe9e63617aafd88403f7fc93b4b305a0c6d603a98a8572e5b5f1",
|
||||||
|
"final_model_state": "3f0b97ece3a15571ba3d656f589f512ca0bb9e20083c9f58a42ccaee14892f59",
|
||||||
|
"final_optimizer_state": "ed03e6fbd4a12d8b063dcb22e0437754285f54d585374cd52fbd534f05d24637"
|
||||||
|
},
|
||||||
|
"2026073002": {
|
||||||
|
"raw_file_sha256": "21199deb2199395061e51220fd8c7afd04a1135a6381e406da9b5795e3ad5032",
|
||||||
|
"canonical_sha256": "b4262e697e2269457fdebf31f75008383c6e8024ee1bbf96e5962fb2ba1dec18",
|
||||||
|
"final_model_state": "bd2556388aeaa211b798c283c7cbd8ccd29edf166a2922fa13d172e8dfdc38d1",
|
||||||
|
"final_optimizer_state": "0b101eab3bc7d8d654be2ea335c86fc25563ce19912d721844ee4e639c569e77"
|
||||||
|
},
|
||||||
|
"2026073003": {
|
||||||
|
"raw_file_sha256": "c0d7f1bcfa9fa7f8f3134ca4571bdf23a951182d03b7de9611a6b4b89667d4ac",
|
||||||
|
"canonical_sha256": "513b85666fcdbf75424596e968cd73733314ec188345e453998de08b18dcef65",
|
||||||
|
"final_model_state": "638568aede21890773b6932a19ec4e112f5ac0a4770ba3402fcd82980a9ecf76",
|
||||||
|
"final_optimizer_state": "83947fd743ec8e3e31ca7788fd201981846f0f1c7e9e38c88afcef95cfc6ec4e"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,406 @@
|
|||||||
|
# K3 Attention Residuals 局部梯度尖峰与归约敏感性协议
|
||||||
|
|
||||||
|
协议 ID:`llm-atlas-k3-attnres-spike-path-v1`
|
||||||
|
冻结日期:2026-07-30
|
||||||
|
协议状态:**结果前预注册**
|
||||||
|
父协议:`llm-atlas-k3-attnres-gradient-scale-v1`
|
||||||
|
|
||||||
|
## 0. 研究身份
|
||||||
|
|
||||||
|
本轮是 Round 05 的**定向机制追踪**,不是盲发现:
|
||||||
|
|
||||||
|
- 已知 depth-32 / Block 的三 seed 平均 layer 21–25 normalized post-MLP
|
||||||
|
activation-gradient RMS 较高;
|
||||||
|
- 已知 observationally,MLP mixer latest-source mass 与该梯度谱相关;
|
||||||
|
- 未知尖峰最早在哪个 block position 出现;
|
||||||
|
- 未知 mixer 的 softmax/key derivative path 与 learned value coefficients 分别贡献多少;
|
||||||
|
- 未知更换 raw gradient tensor 的公开 reduction 后,layer 21–25 是否仍构成稳定局部峰。
|
||||||
|
|
||||||
|
前置已知结果和相关分析固定在
|
||||||
|
`research/K3_ATTNRES_SPIKE_SCOPING.md`。任何 Round 06 输出不得被倒写成“事前未知”。
|
||||||
|
|
||||||
|
## 1. 允许回答的问题
|
||||||
|
|
||||||
|
1. 在同一个缩小 Block AttnRes 模型里,layer 21–25 的相对高值在哪些
|
||||||
|
attention / MLP 位置已经可见?
|
||||||
|
2. 在前向完全相同的条件下,对**全部 mixer** 切断 weight / key 导数路径是否降低尖峰?
|
||||||
|
3. 在前向完全相同的条件下,把**全部 mixer** 的 source value 反向系数改为均匀权重,
|
||||||
|
是否降低尖峰?
|
||||||
|
4. element RMS、token-RMS mean / median / p95 是否给出一致的 spike 结论?
|
||||||
|
5. Round 06 的新增观测是否保持 Round 05 训练状态、BPC 和 post-MLP global RMS exact?
|
||||||
|
|
||||||
|
## 2. 明确不回答的问题
|
||||||
|
|
||||||
|
- 论文 Figure 5(c) 的未公开 telemetry 定义;
|
||||||
|
- Kimi K3 2.8T checkpoint 的真实梯度;
|
||||||
|
- learned mixer 对大模型最终能力的因果收益;
|
||||||
|
- 改写 backward rule 后重新训练会发生什么;
|
||||||
|
- 哪个 source 具有可命名的语义;
|
||||||
|
- 三个 seed 之外的总体显著性或置信区间;
|
||||||
|
- 同 FLOPs、wall time 或参数量公平性。
|
||||||
|
|
||||||
|
## 3. 冻结训练合同
|
||||||
|
|
||||||
|
Round 06 不创建新训练任务分布,完整复用 Round 05:
|
||||||
|
|
||||||
|
| 字段 | 固定值 |
|
||||||
|
|---|---|
|
||||||
|
| architecture | Block AttnRes |
|
||||||
|
| Transformer depth | 32 |
|
||||||
|
| aggregation groups | 8 |
|
||||||
|
| blocks / group | 4 |
|
||||||
|
| width / heads / FFN | 192 / 6 / 768 |
|
||||||
|
| context / vocabulary | 256 / byte-256 |
|
||||||
|
| formal seeds | 2026073001 / 2026073002 / 2026073003 |
|
||||||
|
| steps / batch | 8,000 / 32 |
|
||||||
|
| target bytes / formal cell | 65,536,000 |
|
||||||
|
| optimizer | AdamW |
|
||||||
|
| peak / min LR | 3e-4 / 3e-5 |
|
||||||
|
| warmup | 400 |
|
||||||
|
| weight decay | 0.1 for ndim ≥ 2 |
|
||||||
|
| betas / epsilon | 0.9, 0.95 / 1e-8 |
|
||||||
|
| clip | global norm 1.0 |
|
||||||
|
| forward | CUDA BF16 autocast |
|
||||||
|
| residual accumulation | explicit FP32 |
|
||||||
|
| diagnostic CE | fixed 16 × 256 token-mean FP32 CE |
|
||||||
|
| diagnostic steps | 0 / 100 / 500 / 2,000 / 4,000 / 8,000 |
|
||||||
|
|
||||||
|
数据 bytes、training schedule、validation tensor 和 diagnostic tensor hashes 必须与父协议
|
||||||
|
manifest exact。训练窗口仍由父协议 ID 派生;Round 06 ID 只标识新增 telemetry,不能改变
|
||||||
|
任一 optimizer input。
|
||||||
|
|
||||||
|
## 4. 正式网格与 replay
|
||||||
|
|
||||||
|
正式运行:
|
||||||
|
|
||||||
|
```text
|
||||||
|
depth-32 / block / seed-2026073001
|
||||||
|
depth-32 / block / seed-2026073002
|
||||||
|
depth-32 / block / seed-2026073003
|
||||||
|
```
|
||||||
|
|
||||||
|
另从初始化完整重跑:
|
||||||
|
|
||||||
|
```text
|
||||||
|
replay / depth-32 / block / seed-2026073001
|
||||||
|
```
|
||||||
|
|
||||||
|
正式三格共处理 196,608,000 target bytes;含 replay 共 262,144,000 bytes。
|
||||||
|
|
||||||
|
每格都必须在全新 Python 进程中运行。可并行两个进程,但不能共享 model、optimizer、RNG
|
||||||
|
或 CUDA graph。性能计时不进入数值复现合同。
|
||||||
|
|
||||||
|
## 5. 六个 activation 位置
|
||||||
|
|
||||||
|
对每个 Transformer block `l` 捕获:
|
||||||
|
|
||||||
|
1. `pre_attention_input`:attention RMSNorm 的输入;Block 中是 attention mixer 输出;
|
||||||
|
2. `attention_branch_output`:attention projection 输出;
|
||||||
|
3. `post_attention_state`:attention branch 加入后的 FP32 partial state;
|
||||||
|
4. `pre_mlp_input`:MLP RMSNorm 的输入;Block 中是独立 MLP mixer 输出;
|
||||||
|
5. `mlp_branch_output`:MLP down projection 输出;
|
||||||
|
6. `post_mlp_state`:MLP branch 加入后的 FP32 partial state,即 Round 05 主对象。
|
||||||
|
|
||||||
|
每个位置必须有 32 个不同计算节点,shape 为 `[16,256,192]`,梯度全部 present / finite。
|
||||||
|
跨位置允许计算图语义上的共享来源,但同一位置的 32 个条目不得意外复用同一 storage。
|
||||||
|
|
||||||
|
捕获在一次 learned-mode diagnostic forward / backward 中完成;不能把六个位置拆成六次
|
||||||
|
不同 loss 的 backward:
|
||||||
|
|
||||||
|
- `attention_branch_output` / `mlp_branch_output` 在原 module 输出、转成 FP32 residual
|
||||||
|
之前捕获,保留其实际 autocast dtype;
|
||||||
|
- `post_attention_state` / `post_mlp_state` 在显式 FP32 residual accumulation 后捕获;
|
||||||
|
- 所有 `retain_grad()` 只允许出现在 diagnostic `capture=True` 路径;
|
||||||
|
- 8,000 个 optimizer training steps 必须走父 runner 的原始 `DepthMixer.forward`,
|
||||||
|
`capture=False`,不得进入 intervention custom autograd;
|
||||||
|
- 每次 diagnostic / intervention 前后都 `zero_grad(set_to_none=True)`;
|
||||||
|
- diagnostic 前后 optimizer-state tensor hash 必须 exact。
|
||||||
|
|
||||||
|
### 5.1 尖峰可见位置
|
||||||
|
|
||||||
|
固定目标集合:
|
||||||
|
|
||||||
|
```text
|
||||||
|
S = layers 21, 22, 23, 24, 25
|
||||||
|
R = other 27 layers
|
||||||
|
```
|
||||||
|
|
||||||
|
对每个位置和 seed:
|
||||||
|
|
||||||
|
```text
|
||||||
|
spike_contrast = mean(metric[S]) / mean(metric[R])
|
||||||
|
peak_normalized = max(metric) / mean(metric)
|
||||||
|
```
|
||||||
|
|
||||||
|
主判定**只使用 step 8,000**。若某个位置 `spike_contrast ≥ 1.5` 在 3 / 3 seed
|
||||||
|
成立,则称“该位置已可见局部峰”。六个位置按上述 forward 顺序报告;第一个满足者只标为
|
||||||
|
**earliest tensor where the pattern is observed**,禁止写成 origin、injection point 或
|
||||||
|
“在该算子生成”。其他 diagnostic steps 只展示轨迹,不参与位置判定。若没有位置 3 / 3
|
||||||
|
达标,结论为 position-mixed。
|
||||||
|
|
||||||
|
## 6. 同一 raw gradient tensor 的 reductions
|
||||||
|
|
||||||
|
令某层某位置梯度为 `g ∈ R[B,T,D]`,其中 `B=16,T=256,D=192`。
|
||||||
|
|
||||||
|
### 6.1 主 sensitivity family
|
||||||
|
|
||||||
|
1. `element_rms = sqrt(mean_btd(g²))`
|
||||||
|
2. `token_rms_mean = mean_bt(sqrt(mean_d(g²)))`
|
||||||
|
3. `token_rms_median = median_bt(sqrt(mean_d(g²)))`
|
||||||
|
4. `token_rms_p95 = p95_bt(sqrt(mean_d(g²)))`
|
||||||
|
|
||||||
|
这四项都测每 Token 梯度长度的分布,只改变平方根与 Token reduction 的次序/统计量。
|
||||||
|
|
||||||
|
median / p95 统一使用排序后的 **Hyndman–Fan Type 7 linear interpolation**:
|
||||||
|
|
||||||
|
```text
|
||||||
|
h = (N - 1) × p
|
||||||
|
j = floor(h)
|
||||||
|
q_p = x_sorted[j] + (h - j) × (x_sorted[j + 1] - x_sorted[j])
|
||||||
|
```
|
||||||
|
|
||||||
|
`p=.5/.95`,零基下标;若 `h` 为整数则直接取 `x_sorted[h]`。实现不得依赖 numpy /
|
||||||
|
torch 版本相关的默认 quantile 方法。
|
||||||
|
|
||||||
|
### 6.2 cancellation-sensitive diagnostics
|
||||||
|
|
||||||
|
5. `batch_mean_rms = sqrt(mean_td((mean_b g)²))`
|
||||||
|
6. `token_mean_rms = sqrt(mean_bd((mean_t g)²))`
|
||||||
|
|
||||||
|
它们允许正负梯度先抵消,测的是更相干的方向信号,只作探索性诊断,不进入主 robustness
|
||||||
|
判定。
|
||||||
|
|
||||||
|
### 6.3 代数控制
|
||||||
|
|
||||||
|
7. `global_l2 = sqrt(sum_btd(g²))`
|
||||||
|
|
||||||
|
固定 shape 下它应满足:
|
||||||
|
|
||||||
|
```text
|
||||||
|
global_l2 = element_rms × sqrt(B×T×D)
|
||||||
|
```
|
||||||
|
|
||||||
|
逐层 normalized spectrum、CV、spike contrast 应与 element RMS 在 `1e-6` 内相同。
|
||||||
|
|
||||||
|
### 6.4 reduction robustness 闸门
|
||||||
|
|
||||||
|
对最终 `post_mlp_state`,每个主 family reduction、每个 seed 必须同时满足:
|
||||||
|
|
||||||
|
1. `spike_contrast ≥ 1.5`;
|
||||||
|
2. 该 reduction 的 top-5 layers 与固定集合 `S` 至少重合 3 层;
|
||||||
|
3. 与 element RMS 的 32-layer Spearman `ρ ≥ 0.8`。
|
||||||
|
|
||||||
|
四种 reductions、三个 seed 全部满足才记为
|
||||||
|
`robust within the preregistered reduction family`。任何一格失败即为 `mixed`;
|
||||||
|
全数不满足才记为 `not robust at this threshold`。不添加事后替代阈值。
|
||||||
|
|
||||||
|
top-5 固定按 `metric descending, layer index ascending` 排序;Spearman 对并列值使用
|
||||||
|
average ranks。cancellation-sensitive diagnostics 不得进入本节判定。
|
||||||
|
|
||||||
|
## 7. mixer backward-path interventions
|
||||||
|
|
||||||
|
三种模式使用完全相同的 learned forward weights 与 model state。intervention 的作用域固定
|
||||||
|
为**全部 64 个 depth mixers 加最终 output mixer**;不允许结果后只改 layer 21–25 邻域。
|
||||||
|
因此本节回答的是“全局改写 mixer backward rule 后,固定尖峰指标是否变化”,不是把某个
|
||||||
|
局部 mixer 宣布为唯一原因。
|
||||||
|
|
||||||
|
### 7.1 `learned`
|
||||||
|
|
||||||
|
原始 mixer:
|
||||||
|
|
||||||
|
```text
|
||||||
|
w = softmax(qᵀ RMSNorm(sources))
|
||||||
|
y = Σ w_i source_i
|
||||||
|
```
|
||||||
|
|
||||||
|
反向同时经过 value coefficients 和 softmax / query / key 路径。
|
||||||
|
|
||||||
|
### 7.2 `detached_learned`
|
||||||
|
|
||||||
|
前向仍用同一 `w`,但 `w` 在反向中 detach:
|
||||||
|
|
||||||
|
```text
|
||||||
|
y = Σ stopgrad(w_i) source_i
|
||||||
|
```
|
||||||
|
|
||||||
|
它保留 learned value coefficients,切断 softmax / query / key 对 source gradient 的路径。
|
||||||
|
|
||||||
|
### 7.3 `uniform_value_backward`
|
||||||
|
|
||||||
|
使用自定义 autograd:
|
||||||
|
|
||||||
|
- 先调用父 runner 原始 `DepthMixer.forward` 得到 `y_parent`;
|
||||||
|
- custom Function 的 forward 直接返回已经计算出的 `y_parent`,不重算或做
|
||||||
|
`y + z - z` 式浮点抵消;
|
||||||
|
- backward 对每个 source 返回 `grad_y / N`;
|
||||||
|
- 不向 learned weights / query / key 回传。
|
||||||
|
|
||||||
|
因此 forward logits、loss、所有 activation 值必须与 `learned` **逐元素 exact**,但 source
|
||||||
|
value 的 backward coefficient 变为均匀。
|
||||||
|
|
||||||
|
`detached_learned` 同样以 custom Function 原样返回 `y_parent`,backward 才按 learned
|
||||||
|
`w_i` 把 `grad_y` 分配给 source。两种 custom Function 都只允许在 step 0 / 8,000
|
||||||
|
diagnostic 使用。
|
||||||
|
|
||||||
|
这些是**全局 backward-rule sensitivity diagnostics**,不是可训练模型变体,也不声称是
|
||||||
|
合理部署方案。
|
||||||
|
|
||||||
|
## 8. intervention 判定
|
||||||
|
|
||||||
|
只在 step 0 和 8,000 运行三模式。
|
||||||
|
|
||||||
|
### 8.1 forward identity gate
|
||||||
|
|
||||||
|
同 seed / step 三模式必须满足:
|
||||||
|
|
||||||
|
- logits tensor SHA-256 exact;
|
||||||
|
- loss FP32 value exact;
|
||||||
|
- 六位置 activation tensor hashes exact;
|
||||||
|
- mixer forward mean / quantile summaries exact。
|
||||||
|
|
||||||
|
任一失败,正式结果无效。
|
||||||
|
|
||||||
|
learned mode 还必须证明:训练和普通 diagnostic 的每个 mixer output 使用父 runner 数值
|
||||||
|
路径;custom Function 不得被训练 step 调用。
|
||||||
|
|
||||||
|
### 8.2 初始化负控制
|
||||||
|
|
||||||
|
step 0 的 mixer query 全为零,learned weights 是均匀分布。三模式的六位置
|
||||||
|
`element_rms`:
|
||||||
|
|
||||||
|
- 32 个 raw values 必须 finite 且严格大于 0;
|
||||||
|
- raw spectra 的逐层相对差必须 ≤ `1e-6`;
|
||||||
|
- normalized spectra 最大绝对差必须 ≤ `1e-6`。
|
||||||
|
|
||||||
|
失败则说明 intervention 实现没有隔离预期路径。
|
||||||
|
|
||||||
|
### 8.3 softmax / key derivative path
|
||||||
|
|
||||||
|
最终 `post_mlp_state`,且 reduction 固定为 `element_rms`:
|
||||||
|
|
||||||
|
```text
|
||||||
|
relative_drop_contrast =
|
||||||
|
(contrast_learned - contrast_detached) / contrast_learned
|
||||||
|
|
||||||
|
relative_drop_peak =
|
||||||
|
(peak_learned - peak_detached) / peak_learned
|
||||||
|
```
|
||||||
|
|
||||||
|
`contrast_learned` 与 `peak_learned` 必须 finite 且 `>1e-30`,否则本格 invalid 并停止
|
||||||
|
聚合。若两项都 `≥20%` 且 3 / 3 seed 同向,记为:
|
||||||
|
|
||||||
|
> 全局切断所有 mixer 的 softmax / query / key source-gradient path,在本阈值下对固定
|
||||||
|
> 尖峰指标有 material sensitivity。
|
||||||
|
|
||||||
|
禁止缩写成“layer 21–25 由 softmax/key path 造成”。
|
||||||
|
|
||||||
|
### 8.4 learned value coefficients
|
||||||
|
|
||||||
|
用 `detached_learned → uniform_value_backward` 的同一公式,分母也必须 finite 且
|
||||||
|
`>1e-30`。若 contrast 和 peak 都下降 `≥20%` 且 3 / 3 seed 同向,记为:
|
||||||
|
|
||||||
|
> 全部 mixer 的 value backward coefficients 从 learned `w` 改成 `1/N`,在本阈值下
|
||||||
|
> 对固定尖峰指标有 material sensitivity。
|
||||||
|
|
||||||
|
`N` 随 source count 变化;均匀系数不等于各 source 的数值贡献均匀。禁止写成“learned
|
||||||
|
value weights 是尖峰唯一原因”。
|
||||||
|
|
||||||
|
未过阈值只表示本协议没有达到“material”规则,不证明路径贡献为零。seed 或两个指标方向
|
||||||
|
分裂时统一记为 `mixed`。
|
||||||
|
|
||||||
|
## 9. mixer 分布摘要
|
||||||
|
|
||||||
|
每个子层 mixer 保存:
|
||||||
|
|
||||||
|
- source count;
|
||||||
|
- source labels;
|
||||||
|
- 每 source mean / p05 / median / p95;
|
||||||
|
- entropy mean;
|
||||||
|
- normalized entropy;
|
||||||
|
- max source mass;
|
||||||
|
- latest-source mass;
|
||||||
|
- group / layer / attention-or-MLP 身份。
|
||||||
|
|
||||||
|
相关分析报告 Pearson 与 Spearman,但永远标为 association;只有第 7–8 节的同前向
|
||||||
|
backward intervention 可以支持路径贡献判断。
|
||||||
|
|
||||||
|
不保存可还原语料内容的逐 Token weight arrays。
|
||||||
|
|
||||||
|
## 10. 训练等价与复现闸门
|
||||||
|
|
||||||
|
每个 formal seed 必须与对应 Round 05 depth-32 / Block raw 结果满足:
|
||||||
|
|
||||||
|
- final model-state hash exact;
|
||||||
|
- final optimizer-state hash exact;
|
||||||
|
- 六个 validation BPC exact;
|
||||||
|
- 六个 Round 05 post-MLP element-RMS arrays exact;
|
||||||
|
- training-history 冻结字段 exact。
|
||||||
|
|
||||||
|
此外,step-0 smoke 对六个位置逐一执行同一 loss 的 `×1 / ×2` backward:
|
||||||
|
|
||||||
|
- 每层每 reduction 中线性尺度应乘 2 的指标,其比值在 `2±1e-5`;
|
||||||
|
- normalized spectrum、CV、spike contrast 的绝对差 ≤ `1e-6`;
|
||||||
|
- shape、dtype、gradient presence / finite 与同位置 storage uniqueness 全部过闸。
|
||||||
|
|
||||||
|
普通 diagnostic 与三种 intervention 前后 optimizer-state hash 必须 exact;训练 step
|
||||||
|
禁止 `capture=True`。
|
||||||
|
|
||||||
|
seed-1 replay 还必须与 Round 06 formal seed-1 的以下字段 exact:
|
||||||
|
|
||||||
|
- 全部训练等价字段;
|
||||||
|
- 六位置、七 reductions、三 intervention 的全部数值;
|
||||||
|
- mixer quantile summaries;
|
||||||
|
- logits / activation tensor hashes;
|
||||||
|
- final model / optimizer hashes。
|
||||||
|
|
||||||
|
排除:
|
||||||
|
|
||||||
|
- run kind / output path;
|
||||||
|
- wall time / step time;
|
||||||
|
- peak allocated / reserved memory;
|
||||||
|
- process / host 瞬时字段。
|
||||||
|
|
||||||
|
任一训练等价字段失败,不能聚合机制结果;replay 失败则公开失败并停止网站结论。
|
||||||
|
|
||||||
|
## 11. 正式工件
|
||||||
|
|
||||||
|
```text
|
||||||
|
experiments/k3/attnres_spike/
|
||||||
|
README.md
|
||||||
|
manifest.json
|
||||||
|
scope.py
|
||||||
|
train.py
|
||||||
|
analyze.py
|
||||||
|
reproduction.json
|
||||||
|
results/raw/*.json
|
||||||
|
|
||||||
|
research/
|
||||||
|
K3_ATTNRES_SPIKE_SCOPING.md
|
||||||
|
K3_ATTNRES_SPIKE_PROTOCOL.md
|
||||||
|
K3_ATTNRES_SPIKE_AUDIT.md
|
||||||
|
|
||||||
|
src/data/
|
||||||
|
k3-attnres-spike.json
|
||||||
|
k3-attnres-spike-compact.json
|
||||||
|
```
|
||||||
|
|
||||||
|
网站必须把四种证据身份分开:
|
||||||
|
|
||||||
|
- Round 05 已知峰;
|
||||||
|
- observational mixer association;
|
||||||
|
- same-forward backward intervention;
|
||||||
|
- reduction robustness / failure。
|
||||||
|
|
||||||
|
## 12. 停止规则
|
||||||
|
|
||||||
|
出现以下任一情况立即停止正式聚合:
|
||||||
|
|
||||||
|
- 训练 hash 不匹配 Round 05;
|
||||||
|
- forward identity gate 失败;
|
||||||
|
- step-0 uniform negative control 失败;
|
||||||
|
- 六位置 loss×2 或 optimizer isolation gate 失败;
|
||||||
|
- activation gradient 缺失、非 finite 或 shape 错误;
|
||||||
|
- replay 数值不 exact;
|
||||||
|
- 正式 raw 文件缺失或 canonical hash 不闭合。
|
||||||
|
|
||||||
|
遇到反结果时不修改集合 `S`、20% 阈值、1.5 contrast、top-5 overlap 或 Spearman 阈值。
|
||||||
@@ -0,0 +1,120 @@
|
|||||||
|
# K3 Attention Residuals 局部梯度尖峰:Round 06 前置定位
|
||||||
|
|
||||||
|
研究日期:2026-07-30
|
||||||
|
阶段身份:**探索性 scoping,不是 Round 06 预注册结果**
|
||||||
|
上游协议:`llm-atlas-k3-attnres-gradient-scale-v1`
|
||||||
|
|
||||||
|
## 1. 为什么先做 scoping
|
||||||
|
|
||||||
|
Round 05 已经公开了一个方向分裂:
|
||||||
|
|
||||||
|
- Block AttnRes 在 depth 16 / 32、三个 seed 的 6 / 6 配对中都改善首/末四分位失衡;
|
||||||
|
- 但它也在 6 / 6 配对中提高全层 activation-gradient CV;
|
||||||
|
- depth 32 的三 seed 平均峰值集中在 layer 21–25。
|
||||||
|
|
||||||
|
这意味着下一步不该再问一个笼统的“梯度是否均匀”,而应该问:
|
||||||
|
|
||||||
|
1. 尖峰在 attention / MLP block 的哪个位置已经出现?
|
||||||
|
2. 它是否与 learned mixer 的 source 权重有关?
|
||||||
|
3. 相关来自 mixer softmax / key 的导数路径,还是来自 learned value coefficients?
|
||||||
|
4. 换一种公开、合理的 gradient reduction 后,尖峰是否还存在?
|
||||||
|
|
||||||
|
Round 05 已保存每个诊断点的逐层 post-MLP gradient RMS 和 64 个子层 mixer
|
||||||
|
分布。这里先用那些**已经见过的数据**定位假设,再冻结新协议。Round 06 的因果式
|
||||||
|
backward-path intervention 不能被描述成盲验证。
|
||||||
|
|
||||||
|
## 2. 对齐规则
|
||||||
|
|
||||||
|
depth 32 的 Block AttnRes 使用 8 个 aggregation groups,每组 4 个 Transformer
|
||||||
|
blocks、8 个 attention / MLP residual sublayers。
|
||||||
|
|
||||||
|
对 1-based Transformer layer `l`:
|
||||||
|
|
||||||
|
```text
|
||||||
|
attention mixer index = 2 × (l - 1)
|
||||||
|
MLP mixer index = 2 × (l - 1) + 1
|
||||||
|
group = floor((l - 1) / 4) + 1
|
||||||
|
offset in group = ((l - 1) mod 4) + 1
|
||||||
|
```
|
||||||
|
|
||||||
|
`latest source mass` 定义为对应 mixer `mean_weights` 的最后一个元素。它在不同
|
||||||
|
group offset 的语义并不完全相同:
|
||||||
|
|
||||||
|
- group 首层 attention mixer 没有 current partial,最后源是最近完成的 group;
|
||||||
|
- 其他 attention mixer 的最后源是 current partial;
|
||||||
|
- 每层 MLP mixer 的最后源都是刚加入 attention branch 后的 current partial。
|
||||||
|
|
||||||
|
因此不能把所有 `last` 自动命名为同一种“最近层贡献”。
|
||||||
|
|
||||||
|
## 3. layer 19–28 的三 seed 均值
|
||||||
|
|
||||||
|
下表的 gradient 是每个 seed 先按该 seed 的 32 层均值归一化,再跨 seed 平均。
|
||||||
|
权重和 normalized entropy 也跨三个 seed 平均。
|
||||||
|
|
||||||
|
| layer | group / offset | gradient / layer mean | attn last | MLP last | attn H/logN | MLP H/logN |
|
||||||
|
|---:|---:|---:|---:|---:|---:|---:|
|
||||||
|
| 19 | 5 / 3 | 1.063 | 0.349 | 0.452 | 0.929 | 0.842 |
|
||||||
|
| 20 | 5 / 4 | 1.014 | 0.316 | 0.419 | 0.946 | 0.863 |
|
||||||
|
| 21 | 6 / 1 | **3.042** | 0.250 | **0.513** | 0.969 | 0.763 |
|
||||||
|
| 22 | 6 / 2 | **2.405** | 0.436 | **0.563** | 0.851 | 0.721 |
|
||||||
|
| 23 | 6 / 3 | **1.910** | 0.400 | **0.551** | 0.879 | 0.737 |
|
||||||
|
| 24 | 6 / 4 | **1.548** | 0.384 | **0.515** | 0.889 | 0.771 |
|
||||||
|
| 25 | 7 / 1 | **1.769** | 0.317 | 0.358 | 0.934 | 0.766 |
|
||||||
|
| 26 | 7 / 2 | 1.369 | 0.343 | 0.420 | 0.876 | 0.764 |
|
||||||
|
| 27 | 7 / 3 | 0.986 | 0.325 | 0.402 | 0.891 | 0.775 |
|
||||||
|
| 28 | 7 / 4 | 0.802 | 0.296 | 0.304 | 0.902 | 0.822 |
|
||||||
|
|
||||||
|
layer 21–24 正好是第 6 个 group,layer 25 是第 7 个 group 的首层。梯度峰并不
|
||||||
|
只是一个 group boundary 单点;它在第 6 组内部递减,并在下一组首层出现较小的第二峰。
|
||||||
|
|
||||||
|
## 4. 相关线索
|
||||||
|
|
||||||
|
把三个 seed 的 32 层合成 96 个点,post-MLP normalized gradient 与 mixer 摘要的
|
||||||
|
Pearson 相关为:
|
||||||
|
|
||||||
|
| 变量 | 96 点 `r` | layer 19–28 的 30 点 `r` |
|
||||||
|
|---|---:|---:|
|
||||||
|
| attention latest-source mass | 0.066 | 0.158 |
|
||||||
|
| MLP latest-source mass | **0.651** | **0.690** |
|
||||||
|
| attention normalized entropy | −0.296 | −0.057 |
|
||||||
|
| MLP normalized entropy | −0.274 | **−0.636** |
|
||||||
|
| attention max source mass | −0.024 | 0.158 |
|
||||||
|
| MLP max source mass | 0.230 | **0.702** |
|
||||||
|
|
||||||
|
这些数值只支持:
|
||||||
|
|
||||||
|
> 尖峰层与更集中的 MLP mixer、尤其较大的 latest-partial mean weight 同时出现。
|
||||||
|
|
||||||
|
它们不支持:
|
||||||
|
|
||||||
|
- “MLP latest weight 导致梯度尖峰”;
|
||||||
|
- “第 6 group 是唯一原因”;
|
||||||
|
- “降低 entropy 就一定增加梯度”;
|
||||||
|
- “K3 真实 checkpoint 也有相同模式”。
|
||||||
|
|
||||||
|
同一个 learned mixer 同时改变 forward activation、value-path gradient coefficient
|
||||||
|
和 softmax/key derivative path。单看 observational correlation 无法分解这三者。
|
||||||
|
|
||||||
|
## 5. Round 06 要冻结的可证伪问题
|
||||||
|
|
||||||
|
Round 06 将复跑与 Round 05 完全相同的 depth-32 / Block 三 seed 训练,并要求最终
|
||||||
|
model / optimizer hash 与 Round 05 exact。新增诊断只在 optimizer step 之外执行。
|
||||||
|
|
||||||
|
它将同时测量:
|
||||||
|
|
||||||
|
- pre-attention input;
|
||||||
|
- attention branch output;
|
||||||
|
- post-attention partial state;
|
||||||
|
- pre-MLP input;
|
||||||
|
- MLP branch output;
|
||||||
|
- post-MLP partial state;
|
||||||
|
- 同一 raw gradient tensor 的多种 reduction;
|
||||||
|
- learned backward;
|
||||||
|
- learned weights detached backward;
|
||||||
|
- learned forward + uniform value backward。
|
||||||
|
|
||||||
|
其中后两项保持**前向 logits、loss 与 activation 完全相同**,只改变反向路径。这样才能
|
||||||
|
区分“相关”与“哪条 backward path 对尖峰有实质贡献”。
|
||||||
|
|
||||||
|
完整阈值、失败规则和复现合同见
|
||||||
|
`research/K3_ATTNRES_SPIKE_PROTOCOL.md`。
|
||||||
Reference in New Issue
Block a user