3276 lines
123 KiB
TypeScript
3276 lines
123 KiB
TypeScript
export type PaperTopic =
|
||
| "基础"
|
||
| "Transformer"
|
||
| "长上下文"
|
||
| "MoE"
|
||
| "Scaling"
|
||
| "数据"
|
||
| "训练系统"
|
||
| "优化器"
|
||
| "低精度"
|
||
| "后训练"
|
||
| "推理"
|
||
| "Agent"
|
||
| "多模态"
|
||
| "推理服务"
|
||
| "评测";
|
||
|
||
export interface Paper {
|
||
year: number;
|
||
title: string;
|
||
url: string;
|
||
topics: PaperTopic[];
|
||
contribution: string;
|
||
spotlight?: "Kimi" | "DeepSeek";
|
||
verified: boolean;
|
||
}
|
||
|
||
export const papers: Paper[] = [
|
||
{
|
||
year: 1948,
|
||
title: "A Mathematical Theory of Communication",
|
||
url: "https://doi.org/10.1002/j.1538-7305.1948.tb01338.x",
|
||
topics: ["基础"],
|
||
contribution: "信息熵与序列概率的理论起点。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 1951,
|
||
title: "Prediction and Entropy of Printed English",
|
||
url: "https://doi.org/10.1002/j.1538-7305.1951.tb01366.x",
|
||
topics: ["基础"],
|
||
contribution: "用已知前文时的下一字母猜测实验估计英语熵与冗余,是 next-symbol 预测的概念先声。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 1953,
|
||
title: "The Population Frequencies of Species and the Estimation of Population Parameters",
|
||
url: "https://doi.org/10.1093/biomet/40.3-4.237",
|
||
topics: ["基础"],
|
||
contribution: "以 frequency of frequencies 估计低频与未见事件的概率质量,奠定 Good–Turing 主线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 1977,
|
||
title: "Perplexity—a measure of the difficulty of speech recognition tasks",
|
||
url: "https://doi.org/10.1121/1.2016299",
|
||
topics: ["基础", "评测"],
|
||
contribution: "早期把 perplexity 用作语音识别语言约束难度的历史节点。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 1987,
|
||
title: "Estimation of Probabilities from Sparse Data for the Language Model Component of a Speech Recognizer",
|
||
url: "https://ieeexplore.ieee.org/document/1165125",
|
||
topics: ["基础"],
|
||
contribution: "用折扣与回退把高阶 N-gram 的未见概率质量交给低阶模型。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 1990,
|
||
title: "Finding Structure in Time",
|
||
url: "https://doi.org/10.1207/s15516709cog1402_1",
|
||
topics: ["基础"],
|
||
contribution: "Elman 网络用循环 context units 在预测任务中形成序列相关的隐藏结构。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 1990,
|
||
title: "Indexing by Latent Semantic Analysis",
|
||
url: "https://doi.org/10.1002/(SICI)1097-4571(199009)41:6%3C391::AID-ASI1%3E3.0.CO;2-9",
|
||
topics: ["基础"],
|
||
contribution: "用词—文档矩阵的低秩分解构造全局潜在语义空间。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 1992,
|
||
title: "Class-Based n-gram Models of Natural Language",
|
||
url: "https://aclanthology.org/J92-4003/",
|
||
topics: ["基础"],
|
||
contribution: "用离散词类在 N-gram 中共享统计,是连续词向量之前的重要参数共享路线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 1994,
|
||
title: "Learning long-term dependencies with gradient descent is difficult",
|
||
url: "https://ieeexplore.ieee.org/document/279181",
|
||
topics: ["基础"],
|
||
contribution: "系统阐明循环网络学习长期存储时的梯度与动力学困难。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 1995,
|
||
title: "Improved Backing-off for M-gram Language Modeling",
|
||
url: "https://doi.org/10.1109/ICASSP.1995.479394",
|
||
topics: ["基础"],
|
||
contribution: "用不同续接上下文的数量构造低阶分布,形成 Kneser–Ney smoothing 的核心直觉。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 1996,
|
||
title: "An Empirical Study of Smoothing Techniques for Language Modeling",
|
||
url: "https://aclanthology.org/P96-1041/",
|
||
topics: ["基础", "评测"],
|
||
contribution: "跨数据规模和模型阶数系统比较平滑方法,建立 modified Kneser–Ney 的实证坐标。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 1997,
|
||
title: "Long Short-Term Memory",
|
||
url: "https://doi.org/10.1162/neco.1997.9.8.1735",
|
||
topics: ["基础"],
|
||
contribution: "用受门控的恒定误差通路缓解循环网络的长期依赖与衰减误差回流。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2001,
|
||
title: "A Bit of Progress in Language Modeling",
|
||
url: "https://arxiv.org/abs/cs/0108005",
|
||
topics: ["基础", "评测"],
|
||
contribution: "组合 cache、高阶 N-gram、skipping、Kneser–Ney 与 clustering,展现计数时代的强系统基线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2003,
|
||
title: "A Neural Probabilistic Language Model",
|
||
url: "https://www.jmlr.org/papers/v3/bengio03a.html",
|
||
topics: ["基础"],
|
||
contribution: "用分布式词表示和神经网络突破 N-gram 稀疏泛化。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2005,
|
||
title: "Hierarchical Probabilistic Neural Network Language Model",
|
||
url: "https://proceedings.mlr.press/r5/morin05a.html",
|
||
topics: ["基础"],
|
||
contribution: "用二叉词树把全词表预测改成路径二分类,直面神经 LM 的 Softmax 词表税。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2010,
|
||
title: "Recurrent neural network based language model",
|
||
url: "https://www.isca-archive.org/interspeech_2010/mikolov10_interspeech.html",
|
||
topics: ["基础", "评测"],
|
||
contribution: "把变长循环状态带入真实语音识别 LM,并明确暴露高训练复杂度。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2013,
|
||
title: "Efficient Estimation of Word Representations in Vector Space",
|
||
url: "https://arxiv.org/abs/1301.3781",
|
||
topics: ["基础"],
|
||
contribution: "Word2Vec 把大规模词向量训练变得简单高效。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2013,
|
||
title: "Distributed Representations of Words and Phrases and their Compositionality",
|
||
url: "https://proceedings.neurips.cc/paper/2013/hash/9aa42b31882ec039965f3c4923ce901b-Abstract.html",
|
||
topics: ["基础"],
|
||
contribution: "以 negative sampling、频繁词子采样和短语发现扩展高效静态词表示。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2013,
|
||
title: "One Billion Word Benchmark for Measuring Progress in Statistical Language Modeling",
|
||
url: "https://arxiv.org/abs/1312.3005",
|
||
topics: ["基础", "评测"],
|
||
contribution: "提供接近十亿词的可比 LM 基准,并并列计数、循环与组合系统。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2014,
|
||
title: "GloVe: Global Vectors for Word Representation",
|
||
url: "https://aclanthology.org/D14-1162/",
|
||
topics: ["基础"],
|
||
contribution: "在全局共现计数上拟合加权 log-bilinear 关系,连接矩阵分解与局部窗口表示。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2014,
|
||
title: "Learning Phrase Representations using RNN Encoder-Decoder for Statistical Machine Translation",
|
||
url: "https://aclanthology.org/D14-1179/",
|
||
topics: ["基础"],
|
||
contribution: "引入 RNN encoder–decoder 与 GRU 门控,把变长源序列映射到条件生成。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2014,
|
||
title: "Sequence to Sequence Learning with Neural Networks",
|
||
url: "https://arxiv.org/abs/1409.3215",
|
||
topics: ["基础"],
|
||
contribution: "Encoder—decoder 统一可变长序列映射,也暴露固定向量瓶颈。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2014,
|
||
title: "Neural Machine Translation by Jointly Learning to Align and Translate",
|
||
url: "https://arxiv.org/abs/1409.0473",
|
||
topics: ["基础", "Transformer"],
|
||
contribution: "用可微软对齐让 decoder 直接检索全部 encoder state。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2015,
|
||
title: "Effective Approaches to Attention-based Neural Machine Translation",
|
||
url: "https://aclanthology.org/D15-1166/",
|
||
topics: ["基础", "Transformer"],
|
||
contribution: "系统化 global/local attention 与多种打分形式,澄清 Attention 前夜的设计空间。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2015,
|
||
title: "Neural Machine Translation of Rare Words with Subword Units",
|
||
url: "https://arxiv.org/abs/1508.07909",
|
||
topics: ["基础"],
|
||
contribution: "把 BPE 引入神经机器翻译,形成现代子词分词主线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2016,
|
||
title: "Character-Aware Neural Language Models",
|
||
url: "https://arxiv.org/abs/1508.06615",
|
||
topics: ["基础"],
|
||
contribution: "用字符 CNN 与 highway network 构造词表示,再交给 LSTM LM 建模。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2016,
|
||
title: "Exploring the Limits of Language Modeling",
|
||
url: "https://arxiv.org/abs/1602.02410",
|
||
topics: ["基础", "训练系统", "评测"],
|
||
contribution: "在 Transformer 前夜把 RNN/CNN 词级语言模型推向大数据、大词表与大计算。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2016,
|
||
title: "Efficient softmax approximation for GPUs",
|
||
url: "https://arxiv.org/abs/1609.04309",
|
||
topics: ["基础", "训练系统"],
|
||
contribution: "Adaptive Softmax 利用词频长尾分配不同计算,降低全词表输出的期望成本。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2016,
|
||
title: "Language Modeling with Gated Convolutional Networks",
|
||
url: "https://arxiv.org/abs/1612.08083",
|
||
topics: ["基础", "训练系统"],
|
||
contribution: "以门控卷积建立可并行的强语言模型路线,说明历史并非 RNN 单线演进。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2017,
|
||
title: "Regularizing and Optimizing LSTM Language Models",
|
||
url: "https://arxiv.org/abs/1708.02182",
|
||
topics: ["基础", "训练系统"],
|
||
contribution: "以 DropConnect 与 NT-ASGD 等配方建立 AWD-LSTM 强基线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 1991,
|
||
title: "Adaptive Mixtures of Local Experts",
|
||
url: "https://doi.org/10.1162/neco.1991.3.1.79",
|
||
topics: ["MoE"],
|
||
contribution: "以门控网络让多个局部专家竞争分工,是现代 MoE 的概念起点。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2015,
|
||
title: "Pointer Networks",
|
||
url: "https://arxiv.org/abs/1506.03134",
|
||
topics: ["Transformer"],
|
||
contribution: "让 Attention 分布直接指向输入位置,展示软路由不仅能混合表示,也能定义可变输出字典。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2016,
|
||
title: "Deep Residual Learning for Image Recognition",
|
||
url: "https://openaccess.thecvf.com/content_cvpr_2016/html/He_Deep_Residual_Learning_CVPR_2016_paper.html",
|
||
topics: ["Transformer"],
|
||
contribution: "以加法捷径学习残差映射,是 Transformer 深度信息高速公路的视觉前史。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2016,
|
||
title: "Layer Normalization",
|
||
url: "https://arxiv.org/abs/1607.06450",
|
||
topics: ["Transformer"],
|
||
contribution: "在单个样本内按层输入统计量归一化,为序列模型提供训练/推理一致的尺度控制。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2017,
|
||
title: "Attention Is All You Need",
|
||
url: "https://arxiv.org/abs/1706.03762",
|
||
topics: ["Transformer"],
|
||
contribution: "用多头自注意力与 FFN 取代循环,开启高度并行预训练。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2018,
|
||
title: "Self-Attention with Relative Position Representations",
|
||
url: "https://aclanthology.org/N18-2074/",
|
||
topics: ["Transformer", "长上下文"],
|
||
contribution: "把相对距离表示引入 Self-Attention,并把方法扩展为 relation-aware attention。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2018,
|
||
title: "SentencePiece: A simple and language independent subword tokenizer",
|
||
url: "https://arxiv.org/abs/1808.06226",
|
||
topics: ["基础"],
|
||
contribution: "直接从原始句子训练语言无关子词模型。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2018,
|
||
title: "Improving Language Understanding by Generative Pre-Training",
|
||
url: "https://cdn.openai.com/research-covers/language-unsupervised/language_understanding_paper.pdf",
|
||
topics: ["Transformer", "Scaling"],
|
||
contribution: "GPT-1 建立 decoder-only 生成式预训练再迁移的路线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2018,
|
||
title: "BERT: Pre-training of Deep Bidirectional Transformers",
|
||
url: "https://arxiv.org/abs/1810.04805",
|
||
topics: ["Transformer"],
|
||
contribution: "双向 masked language modeling 验证大规模预训练表示。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "Language Models are Unsupervised Multitask Learners",
|
||
url: "https://cdn.openai.com/better-language-models/language_models_are_unsupervised_multitask_learners.pdf",
|
||
topics: ["Transformer", "Scaling"],
|
||
contribution: "GPT-2 展示规模扩大后的零样本任务迁移。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer",
|
||
url: "https://arxiv.org/abs/1910.10683",
|
||
topics: ["Transformer", "Scaling", "数据"],
|
||
contribution: "T5 统一 text-to-text 接口,并以 C4 建立现代 Common Crawl 清洗基线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "Unified Language Model Pre-training for Natural Language Understanding and Generation",
|
||
url: "https://arxiv.org/abs/1905.03197",
|
||
topics: ["Transformer"],
|
||
contribution: "用 Attention mask 在同一 Transformer 中切换单向、双向与序列到序列预训练。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "Transformer-XL: Attentive Language Models Beyond a Fixed-Length Context",
|
||
url: "https://arxiv.org/abs/1901.02860",
|
||
topics: ["Transformer", "长上下文"],
|
||
contribution: "跨 segment recurrence 与相对位置,突破固定窗口。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "Root Mean Square Layer Normalization",
|
||
url: "https://arxiv.org/abs/1910.07467",
|
||
topics: ["Transformer"],
|
||
contribution: "不做均值中心化的轻量归一化,成为现代 LLM 常用组件。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "Are Sixteen Heads Really Better than One?",
|
||
url: "https://papers.neurips.cc/paper_files/paper/2019/hash/2c601ad9d2ff9bc8b282670cdd54f69f-Abstract.html",
|
||
topics: ["Transformer", "评测"],
|
||
contribution: "测试时剪枝显示大量 Attention heads 存在冗余,但不证明这些 head 在训练阶段无用。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "What Does BERT Look at? An Analysis of BERT’s Attention",
|
||
url: "https://aclanthology.org/W19-4828/",
|
||
topics: ["Transformer", "评测"],
|
||
contribution: "在 BERT heads 中观察位置、分隔符、句法和共指模式,为后验分析提供系统案例。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "Attention is not Explanation",
|
||
url: "https://aclanthology.org/N19-1357/",
|
||
topics: ["Transformer", "评测"],
|
||
contribution: "展示 Attention 权重与特征重要性可能不相关,且不同权重可产生近似预测,划定解释边界。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "Attention is not not Explanation",
|
||
url: "https://aclanthology.org/D19-1002/",
|
||
topics: ["Transformer", "评测"],
|
||
contribution: "指出解释结论依赖定义、模型整体、基线与测试协议,为 Attention 解释争论补充对照。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "Language Models are Few-Shot Learners",
|
||
url: "https://arxiv.org/abs/2005.14165",
|
||
topics: ["Transformer", "Scaling"],
|
||
contribution: "GPT-3 展示规模扩大后 in-context learning 的通用接口。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "GLU Variants Improve Transformer",
|
||
url: "https://arxiv.org/abs/2002.05202",
|
||
topics: ["Transformer"],
|
||
contribution: "系统比较 GLU 变体,SwiGLU 进入现代 FFN 配方。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "On Layer Normalization in the Transformer Architecture",
|
||
url: "https://proceedings.mlr.press/v119/xiong20b.html",
|
||
topics: ["Transformer", "训练系统"],
|
||
contribution: "从初始化梯度分析 Post-LN 的 warm-up 需求与 Pre-LN 的优化差异。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "NormFormer: Improved Transformer Pretraining with Extra Normalization",
|
||
url: "https://arxiv.org/abs/2110.09456",
|
||
topics: ["Transformer", "训练系统"],
|
||
contribution: "在 Pre-LN block 内增加归一化以改善层间梯度尺度与预训练稳定性。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "Longformer: The Long-Document Transformer",
|
||
url: "https://arxiv.org/abs/2004.05150",
|
||
topics: ["长上下文"],
|
||
contribution: "局部滑窗加少量全局 Token,将长文档 attention 稀疏化。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "Reformer: The Efficient Transformer",
|
||
url: "https://arxiv.org/abs/2001.04451",
|
||
topics: ["长上下文"],
|
||
contribution: "LSH attention 与可逆层降低长序列时间和内存。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "Linformer: Self-Attention with Linear Complexity",
|
||
url: "https://arxiv.org/abs/2006.04768",
|
||
topics: ["长上下文"],
|
||
contribution: "用低秩投影压缩序列维,使 attention 近似线性。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "Transformers are RNNs: Fast Autoregressive Transformers with Linear Attention",
|
||
url: "https://arxiv.org/abs/2006.16236",
|
||
topics: ["长上下文"],
|
||
contribution: "核化 attention 获得线性复杂度与递归推理形式。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "Rethinking Attention with Performers",
|
||
url: "https://arxiv.org/abs/2009.14794",
|
||
topics: ["长上下文"],
|
||
contribution: "用随机特征近似 softmax attention,兼顾无偏与线性复杂度。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "Linear Transformers Are Secretly Fast Weight Programmers",
|
||
url: "https://arxiv.org/abs/2102.11174",
|
||
topics: ["长上下文"],
|
||
contribution: "把线性 attention 解释为 fast weights,并引入 delta update 视角。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "RoFormer: Enhanced Transformer with Rotary Position Embedding",
|
||
url: "https://arxiv.org/abs/2104.09864",
|
||
topics: ["Transformer", "长上下文"],
|
||
contribution: "用旋转把绝对位置注入点积并自然表达相对距离。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "Train Short, Test Long: Attention with Linear Biases Enables Input Length Extrapolation",
|
||
url: "https://arxiv.org/abs/2108.12409",
|
||
topics: ["长上下文"],
|
||
contribution: "ALiBi 用线性距离偏置实现长度外推。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "FlashAttention: Fast and Memory-Efficient Exact Attention",
|
||
url: "https://arxiv.org/abs/2205.14135",
|
||
topics: ["长上下文", "训练系统"],
|
||
contribution: "以 IO-aware tiling 实现精确 attention,不写出完整矩阵。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning",
|
||
url: "https://arxiv.org/abs/2307.08691",
|
||
topics: ["长上下文", "训练系统"],
|
||
contribution: "改进工作分割与并行度,提高训练和 prefill 吞吐。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "Fast Transformer Decoding: One Write-Head is All You Need",
|
||
url: "https://arxiv.org/abs/1911.02150",
|
||
topics: ["长上下文", "训练系统"],
|
||
contribution: "MQA 让 Query heads 共享 K/V,大幅缩小解码缓存。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "GQA: Training Generalized Multi-Query Transformer Models",
|
||
url: "https://arxiv.org/abs/2305.13245",
|
||
topics: ["长上下文", "训练系统"],
|
||
contribution: "在 MHA 表达力与 MQA 缓存效率之间分组折中。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Extending Context Window of Large Language Models via Positional Interpolation",
|
||
url: "https://arxiv.org/abs/2306.15595",
|
||
topics: ["长上下文"],
|
||
contribution: "压缩位置索引,让 RoPE 模型稳定微调到更长窗口。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "YaRN: Efficient Context Window Extension of Large Language Models",
|
||
url: "https://arxiv.org/abs/2309.00071",
|
||
topics: ["长上下文"],
|
||
contribution: "结合频率分区与注意力温度扩展 RoPE 上下文。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Ring Attention with Blockwise Transformers for Near-Infinite Context",
|
||
url: "https://arxiv.org/abs/2310.01889",
|
||
topics: ["长上下文", "训练系统"],
|
||
contribution: "环形传递 KV block,让序列长度随设备数扩展。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Retentive Network: A Successor to Transformer for Large Language Models",
|
||
url: "https://arxiv.org/abs/2307.08621",
|
||
topics: ["长上下文"],
|
||
contribution: "统一并行、递归和 chunkwise retention 计算。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Mamba: Linear-Time Sequence Modeling with Selective State Spaces",
|
||
url: "https://arxiv.org/abs/2312.00752",
|
||
topics: ["长上下文"],
|
||
contribution: "让状态空间参数随输入变化,建立现代线性序列模型路线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Transformers are SSMs: Generalized Models and Efficient Algorithms Through Structured State Space Duality",
|
||
url: "https://arxiv.org/abs/2405.21060",
|
||
topics: ["长上下文"],
|
||
contribution: "用结构化状态空间对偶统一 attention 与 SSM 算法。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Gated Linear Attention Transformers with Hardware-Efficient Training",
|
||
url: "https://arxiv.org/abs/2312.06635",
|
||
topics: ["长上下文", "训练系统"],
|
||
contribution: "数据依赖 gate 与硬件高效 chunkwise 线性 attention。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Parallelizing Linear Transformers with the Delta Rule over Sequence Length",
|
||
url: "https://arxiv.org/abs/2406.06484",
|
||
topics: ["长上下文", "训练系统"],
|
||
contribution: "让 DeltaNet 在序列维并行训练。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Gated Delta Networks: Improving Mamba2 with Delta Rule",
|
||
url: "https://arxiv.org/abs/2412.06464",
|
||
topics: ["长上下文"],
|
||
contribution: "把快速遗忘 gate 与定点纠错 delta rule 合并。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Griffin: Mixing Gated Linear Recurrences with Local Attention",
|
||
url: "https://arxiv.org/abs/2402.19427",
|
||
topics: ["长上下文"],
|
||
contribution: "证明线性递归与局部 attention 的混合架构可扩展。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Jamba: A Hybrid Transformer-Mamba Language Model",
|
||
url: "https://arxiv.org/abs/2403.19887",
|
||
topics: ["长上下文", "MoE"],
|
||
contribution: "把 Transformer、Mamba 与 MoE 组合到生产级开放模型。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Kimi Linear: An Expressive, Efficient Attention Architecture",
|
||
url: "https://arxiv.org/abs/2510.26692",
|
||
topics: ["长上下文"],
|
||
contribution: "提出 KDA 并用 3:1 KDA/MLA 混合补足线性状态瓶颈。",
|
||
spotlight: "Kimi",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2026,
|
||
title: "Attention Residuals",
|
||
url: "https://arxiv.org/abs/2603.15031",
|
||
topics: ["Transformer", "长上下文"],
|
||
contribution: "让每层以 softmax 选择先前深度表示,并给出 Block AttnRes。",
|
||
spotlight: "Kimi",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2017,
|
||
title: "Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer",
|
||
url: "https://arxiv.org/abs/1701.06538",
|
||
topics: ["MoE"],
|
||
contribution: "现代稀疏门控 MoE 的起点,以条件计算扩张容量。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "GShard: Scaling Giant Models with Conditional Computation and Automatic Sharding",
|
||
url: "https://arxiv.org/abs/2006.16668",
|
||
topics: ["MoE", "训练系统"],
|
||
contribution: "把 MoE、自动切分与大规模 TPU 训练连接起来。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "Switch Transformers: Scaling to Trillion Parameter Models with Simple and Efficient Sparsity",
|
||
url: "https://arxiv.org/abs/2101.03961",
|
||
topics: ["MoE"],
|
||
contribution: "用 top-1 路由简化稀疏模型并扩到万亿参数。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "BASE Layers: Simplifying Training of Large, Sparse Models",
|
||
url: "https://arxiv.org/abs/2103.16716",
|
||
topics: ["MoE"],
|
||
contribution: "以线性分配求解专家负载平衡,减少辅助损失。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "GLaM: Efficient Scaling of Language Models with Mixture-of-Experts",
|
||
url: "https://arxiv.org/abs/2112.06905",
|
||
topics: ["MoE", "Scaling", "数据"],
|
||
contribution: "展示 MoE 在能耗与推理计算上高效扩展语言模型。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "ST-MoE: Designing Stable and Transferable Sparse Expert Models",
|
||
url: "https://arxiv.org/abs/2202.08906",
|
||
topics: ["MoE"],
|
||
contribution: "系统化稀疏专家的稳定训练与迁移配方。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Mixture-of-Experts with Expert Choice Routing",
|
||
url: "https://arxiv.org/abs/2202.09368",
|
||
topics: ["MoE"],
|
||
contribution: "把 token 选专家反转为专家选固定容量的 token,直接控制专家负载。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "From Sparse to Soft Mixtures of Experts",
|
||
url: "https://arxiv.org/abs/2308.00951",
|
||
topics: ["MoE"],
|
||
contribution: "用连续 slot 分配替代硬 top-k,探索完全可微的专家混合。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Mixtral of Experts",
|
||
url: "https://arxiv.org/abs/2401.04088",
|
||
topics: ["MoE"],
|
||
contribution: "强开放稀疏模型,普及每层 top-2 experts 的工程实践。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "DeepSeekMoE: Towards Ultimate Expert Specialization in Mixture-of-Experts Language Models",
|
||
url: "https://arxiv.org/abs/2401.06066",
|
||
topics: ["MoE"],
|
||
contribution: "细粒度专家分割与 shared expert isolation。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model",
|
||
url: "https://arxiv.org/abs/2405.04434",
|
||
topics: ["MoE", "长上下文", "数据", "训练系统"],
|
||
contribution: "联合 DeepSeekMoE 与 MLA,并披露 8.1T 中英数据、过滤取舍和训练配方。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Mixture-of-Depths: Dynamically Allocating Compute in Transformer-Based Language Models",
|
||
url: "https://arxiv.org/abs/2404.02258",
|
||
topics: ["MoE", "Transformer"],
|
||
contribution: "把条件计算从专家宽度扩到网络深度,让不同 Token 动态跳过或进入层。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Auxiliary-Loss-Free Load Balancing Strategy for Mixture-of-Experts",
|
||
url: "https://arxiv.org/abs/2408.15664",
|
||
topics: ["MoE"],
|
||
contribution: "以只影响 top-k 选择的 expert bias 调负载,避免均衡辅助损失干扰主梯度。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "DeepSeek-V3 Technical Report",
|
||
url: "https://arxiv.org/abs/2412.19437",
|
||
topics: ["基础", "MoE", "Scaling", "数据", "训练系统", "低精度"],
|
||
contribution: "671B-A37B;FP8、MTP、无辅助损失平衡与 DualPipe。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2026,
|
||
title: "LatentMoE: Toward Optimal Accuracy per FLOP and Parameter in Mixture of Experts",
|
||
url: "https://arxiv.org/abs/2601.18089",
|
||
topics: ["MoE"],
|
||
contribution: "将完整模型宽度与路由专家宽度分离。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "Scaling Laws for Neural Language Models",
|
||
url: "https://arxiv.org/abs/2001.08361",
|
||
topics: ["Scaling"],
|
||
contribution: "建立 loss 与参数、数据、计算之间的幂律经验关系。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Training Compute-Optimal Large Language Models",
|
||
url: "https://arxiv.org/abs/2203.15556",
|
||
topics: ["Scaling"],
|
||
contribution: "Chinchilla 修正参数/Token 配比,强调给定计算下更多数据。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Emergent Abilities of Large Language Models",
|
||
url: "https://arxiv.org/abs/2206.07682",
|
||
topics: ["Scaling", "评测"],
|
||
contribution: "记录若干能力随规模出现非线性跃迁的评测现象。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling",
|
||
url: "https://arxiv.org/abs/2304.01373",
|
||
topics: ["Scaling", "评测"],
|
||
contribution: "公开训练中间 checkpoint,支持研究能力怎样随训练形成。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "LLaMA: Open and Efficient Foundation Language Models",
|
||
url: "https://arxiv.org/abs/2302.13971",
|
||
topics: ["Scaling"],
|
||
contribution: "用更多 Token 训练较小开放模型,推动 compute-efficient 推理。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Llama 2: Open Foundation and Fine-Tuned Chat Models",
|
||
url: "https://arxiv.org/abs/2307.09288",
|
||
topics: ["Scaling", "后训练"],
|
||
contribution: "公开预训练、SFT、RLHF 与安全评测细节。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "The Llama 3 Herd of Models",
|
||
url: "https://arxiv.org/abs/2407.21783",
|
||
topics: ["Scaling", "后训练"],
|
||
contribution: "大规模数据、405B dense、长上下文与多阶段后训练。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "DeepSeek LLM: Scaling Open-Source Language Models with Longtermism",
|
||
url: "https://arxiv.org/abs/2401.02954",
|
||
topics: ["Scaling", "数据"],
|
||
contribution: "中英 dense 基线、跨 91 个 Common Crawl dump 去重与公开 scaling-law 研究。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "The Pile: An 800GB Dataset of Diverse Text for Language Modeling",
|
||
url: "https://arxiv.org/abs/2101.00027",
|
||
topics: ["Scaling", "数据"],
|
||
contribution: "公开多域预训练语料与数据配比研究基础。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "The RefinedWeb Dataset for Falcon LLM",
|
||
url: "https://arxiv.org/abs/2306.01116",
|
||
topics: ["Scaling", "数据"],
|
||
contribution: "证明高质量过滤与去重的 Web 数据可以支撑强模型。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Dolma: an Open Corpus of Three Trillion Tokens for Language Model Pretraining Research",
|
||
url: "https://arxiv.org/abs/2402.00159",
|
||
topics: ["Scaling", "数据"],
|
||
contribution: "开放 3T Token 语料与完整数据处理工具。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "CCNet: Extracting High Quality Monolingual Datasets from Web Crawl Data",
|
||
url: "https://arxiv.org/abs/1911.00359",
|
||
topics: ["数据"],
|
||
contribution: "把去重、语言识别和 Wikipedia-like 质量过滤组织成可扩展网页语料流水线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "Extracting Training Data from Large Language Models",
|
||
url: "https://arxiv.org/abs/2012.07805",
|
||
topics: ["数据", "评测"],
|
||
contribution: "从 GPT-2 抽取逐字训练片段与 PII,实证训练记忆和隐私泄露风险。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "Deduplicating Training Data Makes Language Models Better",
|
||
url: "https://arxiv.org/abs/2107.06499",
|
||
topics: ["数据", "评测"],
|
||
contribution: "连接近重复、验证集重合、训练效率与逐字记忆,建立现代 fuzzy dedup 主线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "ROOTS: One Massive Multilingual Dataset for 1.6TB of Text",
|
||
url: "https://arxiv.org/abs/2303.03915",
|
||
topics: ["数据"],
|
||
contribution: "总结 59 种语言、1.6TB 多语言语料的协作建设、治理与数据文档。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Efficient Training of Language Models to Fill in the Middle",
|
||
url: "https://arxiv.org/abs/2207.14255",
|
||
topics: ["数据"],
|
||
contribution: "通过 PSM/SPM 数据重排获得代码中间补全能力,并消融变换率和格式。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "DoReMi: Optimizing Data Mixtures Speeds Up Language Model Pretraining",
|
||
url: "https://arxiv.org/abs/2305.10429",
|
||
topics: ["数据", "Scaling"],
|
||
contribution: "用小代理模型学习 domain weights,再迁移到更大目标模型。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "SemDeDup: Data-efficient Learning at Web-scale through Semantic Deduplication",
|
||
url: "https://arxiv.org/abs/2303.09540",
|
||
topics: ["数据"],
|
||
contribution: "在预训练 embedding 空间聚类语义近重复,研究硬删除与多样性收益。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "D4: Improving LLM Pretraining via Document De-Duplication and Diversification",
|
||
url: "https://arxiv.org/abs/2308.12284",
|
||
topics: ["数据"],
|
||
contribution: "先去重再主动保留多样性,说明智能重复与随机重复并不等价。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "PALOMA: A Benchmark for Evaluating Language Model Fit",
|
||
url: "https://arxiv.org/abs/2312.10523",
|
||
topics: ["数据", "评测"],
|
||
contribution: "提出固定训练顺序、子文档去污染、细粒度 domain 和 bits-per-byte 语料评测协议。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "The FineWeb Datasets: Decanting the Web for the Finest Text Data at Scale",
|
||
url: "https://arxiv.org/abs/2406.17557",
|
||
topics: ["数据"],
|
||
contribution: "以大规模网页清洗消融构建 FineWeb 与教育价值子集。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "DataComp-LM: In Search of the Next Generation of Language Model Pretraining Datasets",
|
||
url: "https://arxiv.org/abs/2406.11794",
|
||
topics: ["数据", "Scaling", "评测"],
|
||
contribution: "固定 240T candidate pool、模型尺度与 53 项评测,使数据 pipeline 可以同协议比较。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Rephrasing the Web: A Recipe for Compute and Data-Efficient Language Modeling",
|
||
url: "https://arxiv.org/abs/2401.16380",
|
||
topics: ["数据"],
|
||
contribution: "WRAP 用受控改写增加网页语料的表达覆盖,并研究 Token utility。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Fewer Truncations Improve Language Modeling",
|
||
url: "https://arxiv.org/abs/2404.10830",
|
||
topics: ["数据", "训练系统"],
|
||
contribution: "Best-fit packing 减少长文档截断,在固定序列长度下保护文档完整性。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "SoftDedup: Efficient Data Reweighting at Scale",
|
||
url: "https://arxiv.org/abs/2407.06654",
|
||
topics: ["数据"],
|
||
contribution: "按常见 n-gram 频率降低样本权重,探索比硬删除更连续的去重策略。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "MATES: Model-Aware Data Selection for Efficient Language Model Pretraining",
|
||
url: "https://arxiv.org/abs/2406.06046",
|
||
topics: ["数据", "Scaling"],
|
||
contribution: "以当前模型状态动态估计样本对目标验证分布的影响。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Investigating Data Contamination in Modern Benchmarks for Large Language Models",
|
||
url: "https://arxiv.org/abs/2401.06059",
|
||
topics: ["数据", "评测"],
|
||
contribution: "区分 text contamination 与 ground-truth contamination,并检查 n-gram 检测的边界。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "MM1: Methods, Analysis & Insights from Multimodal LLM Pre-training",
|
||
url: "https://arxiv.org/abs/2403.09611",
|
||
topics: ["数据", "多模态"],
|
||
contribution: "系统消融 caption、interleaved image-text、text-only mixture 与视觉组件。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "JEST: Joint Example Selection for Multimodal Learning",
|
||
url: "https://arxiv.org/abs/2406.17711",
|
||
topics: ["数据", "多模态"],
|
||
contribution: "按 batch 的联合可学习性选择多模态样本,研究动态数据价值。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Group-MATES: Data-Efficient Pretraining with Group-Level Data Influence Modeling",
|
||
url: "https://arxiv.org/abs/2502.14709",
|
||
topics: ["数据", "Scaling"],
|
||
contribution: "把模型感知的数据影响估计扩展到 group-level 选择。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Swallow: Continued Pre-training with Synthetic Data for Efficient Language Model Adaptation",
|
||
url: "https://arxiv.org/abs/2505.02881",
|
||
topics: ["数据", "后训练"],
|
||
contribution: "用 SwallowCode 与 SwallowMath 研究受控合成数据的继续预训练收益与边界。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2017,
|
||
title: "Deep Learning Scaling is Predictable, Empirically",
|
||
url: "https://arxiv.org/abs/1712.00409",
|
||
topics: ["Scaling"],
|
||
contribution: "跨语言、视觉、语音与翻译任务观察数据规模和误差之间的可预测幂律。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2018,
|
||
title: "An Empirical Model of Large-Batch Training",
|
||
url: "https://arxiv.org/abs/1812.06162",
|
||
topics: ["Scaling", "训练系统"],
|
||
contribution: "以 gradient noise scale 连接关键 batch size、优化步数和训练计算效率。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "Scaling Laws for Autoregressive Generative Modeling",
|
||
url: "https://arxiv.org/abs/2010.14701",
|
||
topics: ["Scaling", "多模态"],
|
||
contribution: "把自回归幂律扩展到图像、视频、数学和语言,显示指数依赖数据模态。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "Scaling Language Models: Methods, Analysis & Insights from Training Gopher",
|
||
url: "https://arxiv.org/abs/2112.11446",
|
||
topics: ["Scaling", "评测"],
|
||
contribution: "以 280B Gopher 系列研究规模、任务表现、训练实践与能力风险的分化。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Broken Neural Scaling Laws",
|
||
url: "https://arxiv.org/abs/2210.14891",
|
||
topics: ["Scaling"],
|
||
contribution: "用平滑折断幂律刻画平台、拐点、double descent 与斜率改变。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Are Emergent Abilities of Large Language Models a Mirage?",
|
||
url: "https://arxiv.org/abs/2304.15004",
|
||
topics: ["Scaling", "评测"],
|
||
contribution: "展示不连续指标可把平滑的底层改进显示成突然涌现,要求同时检查评分函数。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Scaling Data-Constrained Language Models",
|
||
url: "https://arxiv.org/abs/2305.16264",
|
||
topics: ["Scaling"],
|
||
contribution: "区分唯一数据量与重复看到的 Token,系统研究数据受限时的多轮训练。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Beyond Chinchilla-Optimal: Accounting for Inference in Language Model Scaling Laws",
|
||
url: "https://arxiv.org/abs/2401.00448",
|
||
topics: ["Scaling", "训练系统"],
|
||
contribution: "把推理需求加入训练目标,说明生命周期最优模型可能小于纯训练计算最优模型。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Language Models Scale Reliably With Over-Training and on Downstream Tasks",
|
||
url: "https://arxiv.org/abs/2403.08540",
|
||
topics: ["Scaling", "评测"],
|
||
contribution: "检验超过 Chinchilla 配比后的 loss 与下游趋势,并给出多项可预测缩放结果。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Chinchilla Scaling: A Replication Attempt",
|
||
url: "https://arxiv.org/abs/2404.10102",
|
||
topics: ["Scaling"],
|
||
contribution: "复查原论文三种估计方法,质疑部分精确参数与置信区间但未推翻近似等比分配结论。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "More Compute Is What You Need",
|
||
url: "https://arxiv.org/abs/2404.19484",
|
||
topics: ["Scaling"],
|
||
contribution: "提出训练计算分配会改变可迁移性的争议性假说;在课程中仅作为待检验观点。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Resolving Discrepancies in Compute-Optimal Scaling of Language Models",
|
||
url: "https://arxiv.org/abs/2406.19146",
|
||
topics: ["Scaling"],
|
||
contribution: "从 FLOPs 口径、warmup 与随规模调参解释 Kaplan—Chinchilla 配比差异。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Optimization Hyper-parameter Laws for Large Language Models",
|
||
url: "https://arxiv.org/abs/2409.04777",
|
||
topics: ["Scaling", "训练系统"],
|
||
contribution: "研究最优学习率等优化超参数怎样随模型规模和训练长度变化。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Establishing Task Scaling Laws via Compute-Efficient Model Ladders",
|
||
url: "https://arxiv.org/abs/2412.04403",
|
||
topics: ["Scaling", "评测"],
|
||
contribution: "先由模型/数据预测任务 loss,再由任务 loss 预测能力,以小模型梯队降低外推成本。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Overtrained Language Models Are Harder to Fine-Tune",
|
||
url: "https://arxiv.org/abs/2503.19206",
|
||
topics: ["Scaling", "后训练"],
|
||
contribution: "显示更长预训练可能降低后续微调可塑性,分开 base loss 与最终产品目标。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2026,
|
||
title: "Data-Constrained Language Model Pretraining: Improved Regularization and Scaling Laws",
|
||
url: "https://arxiv.org/abs/2606.06888",
|
||
topics: ["Scaling"],
|
||
contribution: "以 SoftQ 和 masked-input regularization 更新有限唯一数据下的重复训练研究。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2014,
|
||
title: "Adam: A Method for Stochastic Optimization",
|
||
url: "https://arxiv.org/abs/1412.6980",
|
||
topics: ["训练系统", "优化器"],
|
||
contribution: "自适应一阶/二阶矩估计,成为 LLM 优化器基线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2017,
|
||
title: "Decoupled Weight Decay Regularization",
|
||
url: "https://arxiv.org/abs/1711.05101",
|
||
topics: ["训练系统", "优化器"],
|
||
contribution: "AdamW 将权重衰减与梯度更新正确解耦。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2012,
|
||
title: "Large Scale Distributed Deep Networks",
|
||
url: "https://arxiv.org/abs/1206.5533",
|
||
topics: ["训练系统"],
|
||
contribution: "DistBelief 与参数服务器奠定早期大规模分布式深度学习系统路线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2014,
|
||
title: "One Weird Trick for Parallelizing Convolutional Neural Networks",
|
||
url: "https://arxiv.org/abs/1404.5997",
|
||
topics: ["训练系统"],
|
||
contribution: "系统比较数据并行、模型并行及其通信边界。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2016,
|
||
title: "Training Deep Nets with Sublinear Memory Cost",
|
||
url: "https://arxiv.org/abs/1604.06174",
|
||
topics: ["训练系统"],
|
||
contribution: "用 activation checkpointing 以额外计算换取 O(√n) feature-map memory。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2017,
|
||
title: "Mixed Precision Training",
|
||
url: "https://arxiv.org/abs/1710.03740",
|
||
topics: ["低精度", "训练系统"],
|
||
contribution: "建立 FP32 master weights、loss scaling 与 FP32 accumulation 配方。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2018,
|
||
title: "PipeDream: Fast and Efficient Pipeline Parallel DNN Training",
|
||
url: "https://arxiv.org/abs/1806.03377",
|
||
topics: ["训练系统"],
|
||
contribution: "以 1F1B、异步 pipeline 与 weight stashing 提高 stage 利用率。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2018,
|
||
title: "Mesh-TensorFlow: Deep Learning for Supercomputers",
|
||
url: "https://arxiv.org/abs/1811.02084",
|
||
topics: ["训练系统"],
|
||
contribution: "把 tensor dimension 的布局声明映射到逻辑 device mesh。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "GPipe: Efficient Training of Giant Neural Networks using Pipeline Parallelism",
|
||
url: "https://arxiv.org/abs/1811.06965",
|
||
topics: ["训练系统"],
|
||
contribution: "用 micro-batch pipeline 把巨型网络跨设备切层。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism",
|
||
url: "https://arxiv.org/abs/1909.08053",
|
||
topics: ["训练系统"],
|
||
contribution: "建立 Transformer tensor parallel 的高效切分方式。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "ZeRO: Memory Optimizations Toward Training Trillion Parameter Models",
|
||
url: "https://arxiv.org/abs/1910.02054",
|
||
topics: ["训练系统"],
|
||
contribution: "分片优化器、梯度与参数,消除数据并行状态冗余。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "Efficient Large-Scale Language Model Training on GPU Clusters Using Megatron-LM",
|
||
url: "https://arxiv.org/abs/2104.04473",
|
||
topics: ["训练系统"],
|
||
contribution: "组合 TP、PP、DP,并提出 interleaved pipeline schedule。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "ZeRO-Infinity: Breaking the GPU Memory Wall for Extreme Scale Deep Learning",
|
||
url: "https://arxiv.org/abs/2104.07857",
|
||
topics: ["训练系统"],
|
||
contribution: "把 GPU、CPU 与 NVMe 组成可带宽感知的异构训练内存层。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "GSPMD: General and Scalable Parallelization for ML Computation Graphs",
|
||
url: "https://arxiv.org/abs/2105.04663",
|
||
topics: ["训练系统"],
|
||
contribution: "用统一 SPMD 表达和 sharding propagation 扩展模型并行。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "Chimera: Efficiently Training Large-Scale Neural Networks with Bidirectional Pipelines",
|
||
url: "https://arxiv.org/abs/2107.06925",
|
||
topics: ["训练系统"],
|
||
contribution: "从 pipeline 两端注入 micro-batches,降低同步流水线气泡。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "DeepSpeed-MoE: Advancing Mixture-of-Experts Inference and Training",
|
||
url: "https://arxiv.org/abs/2201.05596",
|
||
topics: ["MoE", "训练系统"],
|
||
contribution: "组合 data、expert 与 tensor parallel,系统优化 MoE 训练和推理。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Alpa: Automating Inter- and Intra-Operator Parallelism for Distributed Deep Learning",
|
||
url: "https://arxiv.org/abs/2201.12023",
|
||
topics: ["训练系统"],
|
||
contribution: "用代价模型联合搜索 operator 内和 operator 间并行。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Reducing Activation Recomputation in Large Transformer Models",
|
||
url: "https://arxiv.org/abs/2205.05198",
|
||
topics: ["训练系统"],
|
||
contribution: "以 Megatron sequence parallel 与 selective recomputation 降低激活内存和重算开销。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Tutel: Adaptive Mixture-of-Experts at Scale",
|
||
url: "https://arxiv.org/abs/2206.03382",
|
||
topics: ["MoE", "训练系统"],
|
||
contribution: "让 MoE parallelism、dispatch 和 kernels 自适应模型与硬件配置。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "MegaBlocks: Efficient Sparse Training with Mixture-of-Experts",
|
||
url: "https://arxiv.org/abs/2211.15841",
|
||
topics: ["MoE", "训练系统"],
|
||
contribution: "用 block-sparse kernels 支持 dropless 动态 expert workloads。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "PyTorch FSDP: Experiences on Scaling Fully Sharded Data Parallel",
|
||
url: "https://arxiv.org/abs/2304.11277",
|
||
topics: ["训练系统"],
|
||
contribution: "总结 fully sharded 参数、梯度、状态与 prefetch 的工程实践。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "DeepSpeed Ulysses: System Optimizations for Enabling Training of Extreme Long Sequence Transformer Models",
|
||
url: "https://arxiv.org/abs/2309.14509",
|
||
topics: ["训练系统", "长上下文"],
|
||
contribution: "沿 attention head 做 sequence parallel,扩展极长序列训练。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Zero Bubble Pipeline Parallelism",
|
||
url: "https://arxiv.org/abs/2401.10241",
|
||
topics: ["训练系统"],
|
||
contribution: "拆分 input-gradient 与 weight-gradient,并在内存约束下搜索同步零气泡调度。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "MegaScale: Scaling Large Language Model Training to More Than 10,000 GPUs",
|
||
url: "https://arxiv.org/abs/2402.15627",
|
||
topics: ["训练系统"],
|
||
contribution: "系统讨论 10K+ GPU 集群的并行、网络优化与可靠性。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "USP: A Unified Sequence Parallelism Approach for Long Context Generative AI",
|
||
url: "https://arxiv.org/abs/2405.07719",
|
||
topics: ["长上下文", "训练系统"],
|
||
contribution: "在二维 mesh 上组合 Ulysses All-to-All 与 Ring P2P。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "DeepEP: An Efficient Expert-Parallel Communication Library",
|
||
url: "https://github.com/deepseek-ai/DeepEP",
|
||
topics: ["MoE", "训练系统"],
|
||
contribution: "提供面向 NVLink/RDMA 的高吞吐与低延迟 expert dispatch/combine kernels。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving",
|
||
url: "https://arxiv.org/abs/2407.00079",
|
||
topics: ["训练系统", "长上下文"],
|
||
contribution: "以 KV Cache 为中心解耦 prefill/decode 资源。",
|
||
spotlight: "Kimi",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Muon: An Optimizer for Hidden Layers in Neural Networks",
|
||
url: "https://kellerjordan.github.io/posts/muon/",
|
||
topics: ["优化器"],
|
||
contribution: "把 momentum matrix 经 Newton–Schulz 近似正交化,给出 Muon 的原始定义与参考实现。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Muon is Scalable for LLM Training",
|
||
url: "https://arxiv.org/abs/2502.16982",
|
||
topics: ["训练系统", "优化器"],
|
||
contribution: "补足 Muon 在 LLM 规模的权重衰减与更新尺度配方。",
|
||
spotlight: "Kimi",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "FP8 Formats for Deep Learning",
|
||
url: "https://arxiv.org/abs/2209.05433",
|
||
topics: ["低精度", "训练系统"],
|
||
contribution: "定义适合训练/推理的 E4M3 与 E5M2 FP8 格式。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Microscaling Data Formats for Deep Learning",
|
||
url: "https://arxiv.org/abs/2310.10537",
|
||
topics: ["低精度"],
|
||
contribution: "以 block 共享 scale 支持 MXFP4/MXFP8 等 microscaling 格式。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale",
|
||
url: "https://arxiv.org/abs/2208.07339",
|
||
topics: ["低精度"],
|
||
contribution: "分离离群通道,在保持精度下做 8-bit 推理。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "SmoothQuant: Accurate and Efficient Post-Training Quantization for Large Language Models",
|
||
url: "https://arxiv.org/abs/2211.10438",
|
||
topics: ["低精度"],
|
||
contribution: "把 activation 难量化程度离线迁移到权重。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers",
|
||
url: "https://arxiv.org/abs/2210.17323",
|
||
topics: ["低精度"],
|
||
contribution: "基于近似二阶信息逐层量化大模型权重。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2015,
|
||
title: "Deep Learning with Limited Numerical Precision",
|
||
url: "https://arxiv.org/abs/1502.02551",
|
||
topics: ["低精度"],
|
||
contribution: "系统研究定点低精度训练,并用 stochastic rounding 保住微小更新的长期积累。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2017,
|
||
title: "Quantization and Training of Neural Networks for Efficient Integer-Arithmetic-Only Inference",
|
||
url: "https://arxiv.org/abs/1712.05877",
|
||
topics: ["低精度"],
|
||
contribution: "建立 fake quantization 与量化感知训练的经典整数推理配方。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2018,
|
||
title: "Shampoo: Preconditioned Stochastic Tensor Optimization",
|
||
url: "https://arxiv.org/abs/1802.09568",
|
||
topics: ["优化器"],
|
||
contribution: "沿 tensor 各 mode 构造结构化预条件器,在逐坐标自适应之外利用矩阵几何。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2018,
|
||
title: "Adafactor: Adaptive Learning Rates with Sublinear Memory Cost",
|
||
url: "https://arxiv.org/abs/1804.04235",
|
||
topics: ["优化器", "训练系统"],
|
||
contribution: "用行列因子近似二阶矩,并以 update clipping 降低优化器状态与失稳风险。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "A Study of BFLOAT16 for Deep Learning Training",
|
||
url: "https://arxiv.org/abs/1905.12322",
|
||
topics: ["低精度", "训练系统"],
|
||
contribution: "解释 BF16 保留 FP32 exponent、压缩 mantissa 的范围—精度取舍。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "Query-Key Normalization for Transformers",
|
||
url: "https://arxiv.org/abs/2010.04245",
|
||
topics: ["Transformer", "优化器"],
|
||
contribution: "归一化每个 attention head 的 Q/K 并学习缩放,直接约束 Softmax logit 尺度。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "8-bit Optimizers via Block-wise Quantization",
|
||
url: "https://arxiv.org/abs/2110.02861",
|
||
topics: ["优化器", "低精度"],
|
||
contribution: "用 block-wise dynamic quantization 压缩 optimizer state,同时隔离离群值影响。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "DeepNet: Scaling Transformers to 1,000 Layers",
|
||
url: "https://arxiv.org/abs/2203.00555",
|
||
topics: ["Transformer", "优化器"],
|
||
contribution: "用 DeepNorm 的残差缩放与初始化界约束深层 Transformer 更新。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Tensor Programs V: Tuning Large Neural Networks via Zero-Shot Hyperparameter Transfer",
|
||
url: "https://arxiv.org/abs/2203.03466",
|
||
topics: ["Scaling", "优化器"],
|
||
contribution: "μP 让小代理模型调出的超参数可随宽度迁移到更大网络。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Scaling Vision Transformers to 22 Billion Parameters",
|
||
url: "https://arxiv.org/abs/2302.05442",
|
||
topics: ["Transformer", "优化器", "多模态"],
|
||
contribution: "以 QK LayerNorm 等改动缓解超大 ViT 的 attention logit 爆炸。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Stable and Low-precision Training for Large-scale Vision-language Models",
|
||
url: "https://arxiv.org/abs/2304.13013",
|
||
topics: ["低精度", "优化器", "多模态"],
|
||
contribution: "用 SwitchBack 与稳定性诊断研究低精度 VLM 训练中的 outlier 和 Adam second-moment spike。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "QLoRA: Efficient Finetuning of Quantized LLMs",
|
||
url: "https://arxiv.org/abs/2305.14314",
|
||
topics: ["低精度", "后训练"],
|
||
contribution: "以 NF4、double quantization 和 paged optimizers 微调冻结的 4-bit 基座。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "LLM-QAT: Data-Free Quantization Aware Training for Large Language Models",
|
||
url: "https://arxiv.org/abs/2305.17888",
|
||
topics: ["低精度", "后训练"],
|
||
contribution: "用自生成数据进行 weight、activation 与 KV cache 的量化感知训练。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Small-scale Proxies for Large-scale Transformer Training Instabilities",
|
||
url: "https://arxiv.org/abs/2309.14322",
|
||
topics: ["优化器", "训练系统"],
|
||
contribution: "用高学习率小模型代理复现大模型失稳,降低稳定性机制的研究成本。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "FP8-LM: Training FP8 Large Language Models",
|
||
url: "https://arxiv.org/abs/2310.18313",
|
||
topics: ["低精度", "训练系统"],
|
||
contribution: "把 FP8 扩展到计算、优化器与分布式通信的端到端 LLM 训练。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Spike No More: Stabilizing the Pre-training of Large Language Models",
|
||
url: "https://arxiv.org/abs/2312.16903",
|
||
topics: ["优化器", "Transformer"],
|
||
contribution: "从“小 sublayer、大 shortcut”原则解释并抑制预训练早期 loss spike。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Deconstructing What Makes a Good Optimizer for Language Models",
|
||
url: "https://arxiv.org/abs/2407.07972",
|
||
topics: ["优化器"],
|
||
contribution: "拆解 Adam、Adafactor、Lion 等方法的更新尺度、方向与超参边界。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Scaling FP8 Training to Trillion-Token LLMs",
|
||
url: "https://arxiv.org/abs/2409.12517",
|
||
topics: ["低精度", "训练系统"],
|
||
contribution: "研究万亿 Token 长程 FP8 训练的 SwiGLU outlier,并提出 Smooth-SwiGLU。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "FP4 All the Way: Fully Quantized Training of LLMs",
|
||
url: "https://arxiv.org/abs/2505.19115",
|
||
topics: ["低精度", "训练系统"],
|
||
contribution: "探索权重、激活与梯度广泛采用 FP4 的 fully quantized 训练边界。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2026,
|
||
title: "PowLU: Power-Law Units for Stable and Efficient LLM Training",
|
||
url: "https://arxiv.org/abs/2605.25704",
|
||
topics: ["优化器", "低精度", "Transformer"],
|
||
contribution: "用有界幂律激活控制极值,连接 activation 设计与低精度稳定性。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "AWQ: Activation-aware Weight Quantization for LLM Compression and Acceleration",
|
||
url: "https://arxiv.org/abs/2306.00978",
|
||
topics: ["低精度"],
|
||
contribution: "保护少量高影响权重通道的 activation-aware 量化。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2017,
|
||
title: "Proximal Policy Optimization Algorithms",
|
||
url: "https://arxiv.org/abs/1707.06347",
|
||
topics: ["后训练"],
|
||
contribution: "用 clipped ratio 稳定 on-policy 策略更新。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "Learning to Summarize with Human Feedback",
|
||
url: "https://arxiv.org/abs/2009.01325",
|
||
topics: ["后训练"],
|
||
contribution: "把人类偏好、奖励模型与策略优化应用到生成任务。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "Finetuned Language Models Are Zero-Shot Learners",
|
||
url: "https://arxiv.org/abs/2109.01652",
|
||
topics: ["后训练"],
|
||
contribution: "FLAN 证明多任务指令微调可提升未见任务零样本表现。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "Multitask Prompted Training Enables Zero-Shot Task Generalization",
|
||
url: "https://arxiv.org/abs/2110.08207",
|
||
topics: ["后训练"],
|
||
contribution: "T0 用大规模提示模板训练开放零样本模型。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Training Language Models to Follow Instructions with Human Feedback",
|
||
url: "https://arxiv.org/abs/2203.02155",
|
||
topics: ["后训练"],
|
||
contribution: "InstructGPT 建立 SFT → reward model → PPO 的 RLHF 标准管线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Constitutional AI: Harmlessness from AI Feedback",
|
||
url: "https://arxiv.org/abs/2212.08073",
|
||
topics: ["后训练"],
|
||
contribution: "用原则、自我批评和 AI feedback 减少直接人工标签。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Self-Instruct: Aligning Language Models with Self-Generated Instructions",
|
||
url: "https://arxiv.org/abs/2212.10560",
|
||
topics: ["后训练"],
|
||
contribution: "从模型自生成、过滤指令数据扩展 instruction tuning。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Direct Preference Optimization: Your Language Model is Secretly a Reward Model",
|
||
url: "https://arxiv.org/abs/2305.18290",
|
||
topics: ["后训练"],
|
||
contribution: "把偏好优化改写为直接分类目标,省去在线 RL loop。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2017,
|
||
title: "Deep Reinforcement Learning from Human Preferences",
|
||
url: "https://proceedings.neurips.cc/paper/2017/hash/d5e2c0adad503c91f91df240d0cd4e49-Abstract.html",
|
||
topics: ["后训练"],
|
||
contribution: "把人类对轨迹片段的成对比较拟合成奖励预测器,再用预测奖励优化策略。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "Fine-Tuning Language Models from Human Preferences",
|
||
url: "https://arxiv.org/abs/1909.08593",
|
||
topics: ["后训练"],
|
||
contribution: "把偏好奖励、reference KL 与 PPO 接到生成式语言模型,是现代语言模型 RLHF 的直接前身。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "Natural Instructions: Benchmarking Generalization to New Tasks from Natural Language Instructions",
|
||
url: "https://arxiv.org/abs/2104.08773",
|
||
topics: ["后训练", "评测"],
|
||
contribution: "把任务定义写成声明式自然语言说明,系统检验模型对未见任务的 instruction generalization。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Training a Helpful and Harmless Assistant with Reinforcement Learning from Human Feedback",
|
||
url: "https://arxiv.org/abs/2204.05862",
|
||
topics: ["后训练"],
|
||
contribution: "把 helpfulness 与 harmlessness 分开收集偏好并研究在线迭代,揭示通用助手对齐的多目标性质。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Super-NaturalInstructions: Generalization via Declarative Instructions on 1600+ NLP Tasks",
|
||
url: "https://arxiv.org/abs/2204.07705",
|
||
topics: ["后训练", "评测"],
|
||
contribution: "把自然语言任务定义扩展到 1,600 余个任务,检验跨任务和跨类型泛化。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Scaling Instruction-Finetuned Language Models",
|
||
url: "https://arxiv.org/abs/2210.11416",
|
||
topics: ["后训练", "Scaling"],
|
||
contribution: "联合扩展 Flan 的任务混合、模型规模与 chain-of-thought 数据,研究 instruction tuning 的规模效应。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Scaling Laws for Reward Model Overoptimization",
|
||
url: "https://arxiv.org/abs/2210.10760",
|
||
topics: ["后训练", "Scaling", "评测"],
|
||
contribution: "在 proxy reward 与 gold reward 的受控设置中量化 Goodhart 曲线,展示继续追高代理奖励何时反而伤害真实目标。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "OpenAssistant Conversations: Democratizing Large Language Model Alignment",
|
||
url: "https://arxiv.org/abs/2304.07327",
|
||
topics: ["后训练", "数据"],
|
||
contribution: "开放多轮对话树、候选排序和标注流程,为可复查的助手后训练数据提供公共基线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "RRHF: Rank Responses to Align Language Models with Human Feedback",
|
||
url: "https://arxiv.org/abs/2304.05302",
|
||
topics: ["后训练"],
|
||
contribution: "用回答序列得分的排序约束让模型学习多来源候选的相对质量。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "SLiC-HF: Sequence Likelihood Calibration with Human Feedback",
|
||
url: "https://arxiv.org/abs/2305.10425",
|
||
topics: ["后训练"],
|
||
contribution: "直接校准 chosen 与 rejected 回答的序列似然,为不运行在线 RL 的偏好训练提供早期路线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "LIMA: Less Is More for Alignment",
|
||
url: "https://arxiv.org/abs/2305.11206",
|
||
topics: ["后训练", "数据"],
|
||
contribution: "用约一千条精选示范检验强基座上少量高质量 SFT 的接口塑形能力,并明确其适用边界。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "AlpacaFarm: A Simulation Framework for Methods that Learn from Human Feedback",
|
||
url: "https://arxiv.org/abs/2305.14387",
|
||
topics: ["后训练", "评测"],
|
||
contribution: "提供模拟偏好、自动评测与多种 RLHF 基线,降低比较后训练方法的实验成本。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "RLAIF: Scaling Reinforcement Learning from Human Feedback with AI Feedback",
|
||
url: "https://arxiv.org/abs/2309.00267",
|
||
topics: ["后训练", "评测"],
|
||
contribution: "在摘要、helpful 与 harmless 场景直接比较 AI 和人工偏好标签,而不是把 AI feedback 当作无价值来源。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "UltraFeedback: Boosting Language Models with Scaled AI Feedback",
|
||
url: "https://arxiv.org/abs/2310.01377",
|
||
topics: ["后训练", "数据"],
|
||
contribution: "规模化生成多候选、多维度 AI 反馈,成为开放偏好训练与直接优化的重要数据来源。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "A General Theoretical Paradigm to Understand Learning from Human Preferences",
|
||
url: "https://arxiv.org/abs/2310.12036",
|
||
topics: ["后训练"],
|
||
contribution: "分析确定性偏好下 DPO 的过拟合边界,并以有限间隔回归提出 IPO。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Zephyr: Direct Distillation of LM Alignment",
|
||
url: "https://arxiv.org/abs/2310.16944",
|
||
topics: ["后训练"],
|
||
contribution: "以蒸馏式 SFT、UltraFeedback AI 偏好与直接偏好优化展示开放模型的完整对齐配方。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "KTO: Model Alignment as Prospect Theoretic Optimization",
|
||
url: "https://arxiv.org/abs/2402.01306",
|
||
topics: ["后训练"],
|
||
contribution: "用前景理论式效用处理 desirable / undesirable 二元反馈,放宽必须提供成对偏好的数据合同。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Back to Basics: Revisiting REINFORCE Style Optimization for Learning from Human Feedback in LLMs",
|
||
url: "https://arxiv.org/abs/2402.14740",
|
||
topics: ["后训练"],
|
||
contribution: "以 leave-one-out baseline 重新建立 critic-free REINFORCE 基线,说明 PPO 不是在线 RLHF 的唯一实现。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "ORPO: Monolithic Preference Optimization without Reference Model",
|
||
url: "https://arxiv.org/abs/2403.07691",
|
||
topics: ["后训练"],
|
||
contribution: "把 SFT 负对数似然与 odds-ratio 偏好惩罚合成单阶段、无独立 reference model 的目标。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "RewardBench: Evaluating Reward Models for Language Modeling",
|
||
url: "https://arxiv.org/abs/2403.13787",
|
||
topics: ["后训练", "评测"],
|
||
contribution: "按 chat、hard、safety、reasoning 与分布外样本直接评估奖励模型的成对排序能力。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "SimPO: Simple Preference Optimization with a Reference-Free Reward",
|
||
url: "https://arxiv.org/abs/2405.14734",
|
||
topics: ["后训练"],
|
||
contribution: "用长度归一的平均 log-prob 与目标 margin 构造 reference-free implicit reward。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Unpacking DPO and PPO: Disentangling Best Practices for Learning from Preference Feedback",
|
||
url: "https://proceedings.neurips.cc/paper_files/paper/2024/hash/404df2480b6eef0486a1679e371894b0-Abstract-Conference.html",
|
||
topics: ["后训练", "评测"],
|
||
contribution: "把 preference data、优化算法、奖励模型和 policy prompts 四个变量拆开,避免把配方差异误判为算法胜负。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Chain-of-Thought Prompting Elicits Reasoning in Large Language Models",
|
||
url: "https://arxiv.org/abs/2201.11903",
|
||
topics: ["推理"],
|
||
contribution: "用中间推理示例显著提升大模型复杂任务表现。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Large Language Models are Zero-Shot Reasoners",
|
||
url: "https://arxiv.org/abs/2205.11916",
|
||
topics: ["推理"],
|
||
contribution: "用统一的 step-by-step 触发语句,在不提供 few-shot rationale 时激发多任务零样本 CoT。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Solving Quantitative Reasoning Problems with Language Models",
|
||
url: "https://arxiv.org/abs/2206.14858",
|
||
topics: ["推理", "Scaling"],
|
||
contribution: "Minerva 以技术内容继续训练语言模型,系统推进数学与科学定量推理。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "Training Verifiers to Solve Math Word Problems",
|
||
url: "https://arxiv.org/abs/2110.14168",
|
||
topics: ["推理", "后训练"],
|
||
contribution: "建立 GSM8K,并系统展示多采样后由学习式 verifier 选择答案的收益。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Self-Consistency Improves Chain of Thought Reasoning in Language Models",
|
||
url: "https://arxiv.org/abs/2203.11171",
|
||
topics: ["推理"],
|
||
contribution: "采样多条推理路径并对最终答案聚合。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Least-to-Most Prompting Enables Complex Reasoning in Large Language Models",
|
||
url: "https://arxiv.org/abs/2205.10625",
|
||
topics: ["推理"],
|
||
contribution: "先把难题分解成子问题,再按顺序利用已解结果组合答案。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "STaR: Bootstrapping Reasoning With Reasoning",
|
||
url: "https://arxiv.org/abs/2203.14465",
|
||
topics: ["推理"],
|
||
contribution: "迭代生成、筛选并训练成功 rationale。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Solving Math Word Problems With Process- and Outcome-Based Feedback",
|
||
url: "https://arxiv.org/abs/2211.14275",
|
||
topics: ["推理", "后训练"],
|
||
contribution: "在 GSM8K 中比较过程与结果反馈,揭示最终正确率和推理轨迹错误率的不同需求。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Program of Thoughts Prompting: Disentangling Computation from Reasoning for Numerical Reasoning Tasks",
|
||
url: "https://arxiv.org/abs/2211.12588",
|
||
topics: ["推理", "Agent"],
|
||
contribution: "让模型用程序表达推理,把数值计算交给外部执行器。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Self-Refine: Iterative Refinement with Self-Feedback",
|
||
url: "https://arxiv.org/abs/2303.17651",
|
||
topics: ["推理"],
|
||
contribution: "由同一模型循环生成、反馈与修订,在不追加训练的情况下扩展串行测试时计算。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Tree of Thoughts: Deliberate Problem Solving with Large Language Models",
|
||
url: "https://arxiv.org/abs/2305.10601",
|
||
topics: ["推理", "Agent"],
|
||
contribution: "显式搜索多个 thought 分支并评估中间状态。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Language Models Don't Always Say What They Think: Unfaithful Explanations in Chain-of-Thought Prompting",
|
||
url: "https://arxiv.org/abs/2305.04388",
|
||
topics: ["推理", "评测"],
|
||
contribution: "用偏置提示实验表明,流畅的公开 CoT 可能合理化答案而不忠实披露影响因素。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Let's Verify Step by Step",
|
||
url: "https://arxiv.org/abs/2305.20050",
|
||
topics: ["推理", "后训练"],
|
||
contribution: "发布 PRM800K;在 500 题 MATH 子集的 best-of-1860 选择中验证过程监督优势。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Quiet-STaR: Language Models Can Teach Themselves to Think Before Speaking",
|
||
url: "https://arxiv.org/abs/2403.09629",
|
||
topics: ["推理"],
|
||
contribution: "在一般文本 Token 之间学习隐式 rationale。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models",
|
||
url: "https://arxiv.org/abs/2402.03300",
|
||
topics: ["数据", "推理", "后训练"],
|
||
contribution: "从 40B HTML pages 发现 120.2B 数学语料,并提出无独立 critic 的 GRPO。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Scaling LLM Test-Time Compute Optimally can be More Effective than Scaling Model Parameters",
|
||
url: "https://arxiv.org/abs/2408.03314",
|
||
topics: ["推理", "Scaling"],
|
||
contribution: "按模型与题目难度在修订、并行采样和 PRM 搜索之间分配测试时计算。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Tülu 3: Pushing Frontiers in Open Language Model Post-Training",
|
||
url: "https://arxiv.org/abs/2411.15124",
|
||
topics: ["后训练", "推理"],
|
||
contribution: "开放从数据策展、SFT、偏好学习到 RLVR 的完整 post-training 配方。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "OpenAI o1 System Card",
|
||
url: "https://arxiv.org/abs/2412.16720",
|
||
topics: ["推理", "评测"],
|
||
contribution: "记录 reasoning model 的能力、安全评测与测试时推理边界。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning",
|
||
url: "https://arxiv.org/abs/2501.12948",
|
||
topics: ["推理", "后训练"],
|
||
contribution: "R1-Zero 从 base 直接做规则奖励 RL;R1 再加入冷启动、SFT、多阶段 RL 与蒸馏。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Kimi k1.5: Scaling Reinforcement Learning with LLMs",
|
||
url: "https://arxiv.org/abs/2501.12599",
|
||
topics: ["推理", "后训练"],
|
||
contribution: "用 128K RL context、partial rollout 与 long2short 扩展长 CoT 和测试时计算。",
|
||
spotlight: "Kimi",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "s1: Simple test-time scaling",
|
||
url: "https://arxiv.org/abs/2501.19393",
|
||
topics: ["推理", "后训练"],
|
||
contribution: "从强教师精选 1K 道推理题蒸馏,并用 budget forcing 控制思考长度。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "DAPO: An Open-Source LLM Reinforcement Learning System at Scale",
|
||
url: "https://arxiv.org/abs/2503.14476",
|
||
topics: ["推理", "后训练"],
|
||
contribution: "用 Clip-Higher、Dynamic Sampling、token-level loss 与 overlong shaping 稳定长 CoT RL。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Understanding R1-Zero-Like Training: A Critical Perspective",
|
||
url: "https://arxiv.org/abs/2503.20783",
|
||
topics: ["推理", "后训练"],
|
||
contribution: "提出 Dr.GRPO,分析 response-length 与 question-difficulty 两种优化偏置。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Does Reinforcement Learning Really Incentivize Reasoning Capacity in LLMs Beyond the Base Model?",
|
||
url: "https://arxiv.org/abs/2504.13837",
|
||
topics: ["推理", "后训练", "评测"],
|
||
contribution: "用 pass@k 检查当前 RLVR 是扩展解法覆盖,还是主要重排已有正确路径。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2026,
|
||
title: "MOPD: Multi-Teacher On-Policy Distillation for Capability Integration in LLM Post-Training",
|
||
url: "https://arxiv.org/abs/2606.30406",
|
||
topics: ["推理", "后训练", "Agent"],
|
||
contribution: "让学生在自己的 rollout 上接收同源领域教师的稠密逐 Token 信号,整合多项 RL 能力。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "WebGPT: Browser-assisted Question-Answering with Human Feedback",
|
||
url: "https://arxiv.org/abs/2112.09332",
|
||
topics: ["Agent", "后训练"],
|
||
contribution: "让模型操作浏览器检索、引用来源并用人类偏好训练。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "ReAct: Synergizing Reasoning and Acting in Language Models",
|
||
url: "https://arxiv.org/abs/2210.03629",
|
||
topics: ["Agent", "推理"],
|
||
contribution: "交错 reasoning、action 与 observation,建立 Agent loop 范式。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Toolformer: Language Models Can Teach Themselves to Use Tools",
|
||
url: "https://arxiv.org/abs/2302.04761",
|
||
topics: ["Agent"],
|
||
contribution: "自监督筛选能降低语言建模损失的 API 调用。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Reflexion: Language Agents with Verbal Reinforcement Learning",
|
||
url: "https://arxiv.org/abs/2303.11366",
|
||
topics: ["Agent"],
|
||
contribution: "用语言反思和 episodic memory 改进后续尝试。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Gorilla: Large Language Model Connected with Massive APIs",
|
||
url: "https://arxiv.org/abs/2305.15334",
|
||
topics: ["Agent"],
|
||
contribution: "面向大量真实 API 的工具检索与可靠函数调用。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Voyager: An Open-Ended Embodied Agent with Large Language Models",
|
||
url: "https://arxiv.org/abs/2305.16291",
|
||
topics: ["Agent"],
|
||
contribution: "自动课程、技能库和迭代提示支持开放式长期探索。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "AgentBench: Evaluating LLMs as Agents",
|
||
url: "https://arxiv.org/abs/2308.03688",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "跨操作系统、数据库、游戏等环境统一评估 Agent。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "SWE-bench: Can Language Models Resolve Real-World GitHub Issues?",
|
||
url: "https://arxiv.org/abs/2310.06770",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "用真实仓库 issue 与测试补丁评测软件工程 Agent。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "SWE-agent: Agent-Computer Interfaces Enable Automated Software Engineering",
|
||
url: "https://arxiv.org/abs/2405.15793",
|
||
topics: ["Agent"],
|
||
contribution: "证明专门设计的 Agent-Computer Interface 显著影响解题表现。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments",
|
||
url: "https://arxiv.org/abs/2404.07972",
|
||
topics: ["Agent", "多模态", "评测"],
|
||
contribution: "在真实桌面操作系统中评测视觉 Computer Use Agent。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "τ-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains",
|
||
url: "https://arxiv.org/abs/2406.12045",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "以数据库最终状态和策略约束评估真实工具交互。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "BrowseComp: A Simple Yet Challenging Benchmark for Browsing Agents",
|
||
url: "https://arxiv.org/abs/2504.12516",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "要求网页 Agent 多步寻找难以直接检索的事实。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Kimi K2: Open Agentic Intelligence",
|
||
url: "https://arxiv.org/abs/2507.20534",
|
||
topics: ["MoE", "数据", "训练系统", "Agent", "后训练", "推理"],
|
||
contribution: "开放 1T MoE Agent 模型,以 15.5T、知识/数学改写与 verifiable gym / self-critique joint RL 协同。",
|
||
spotlight: "Kimi",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2026,
|
||
title: "Kimi K2.5: Visual Agentic Intelligence",
|
||
url: "https://arxiv.org/abs/2602.02276",
|
||
topics: ["数据", "训练系统", "Agent", "多模态", "推理"],
|
||
contribution: "把视觉、工具使用和并行 Agent Swarm 纳入统一后训练。",
|
||
spotlight: "Kimi",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2018,
|
||
title: "TextWorld: A Learning Environment for Text-based Games",
|
||
url: "https://arxiv.org/abs/1806.11532",
|
||
topics: ["Agent"],
|
||
contribution: "用可程序生成的文字世界、状态与动作接口,为语言 Agent 建立可控交互环境。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2019,
|
||
title: "Jericho: A Learning Environment for Interactive Fiction Games",
|
||
url: "https://arxiv.org/abs/1909.05398",
|
||
topics: ["Agent"],
|
||
contribution: "把真实交互式小说标准化为含世界状态与语言动作的强化学习环境。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "ALFWorld: Aligning Text and Embodied Environments for Interactive Learning",
|
||
url: "https://arxiv.org/abs/2010.03768",
|
||
topics: ["Agent"],
|
||
contribution: "对齐抽象文本环境与具身家务环境,让文字策略能迁移到更真实的交互世界。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "ScienceWorld: Is your Agent Smarter than a 5th Grader?",
|
||
url: "https://arxiv.org/abs/2203.07540",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "用需要多步实验和科学知识的文本环境评估语言 Agent 的交互推理。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "WebShop: Towards Scalable Real-World Web Interaction with Grounded Language Agents",
|
||
url: "https://arxiv.org/abs/2207.01206",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "以 1.18M 商品、12K 指令和可计算 reward 构造大规模网页购物环境。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Do As I Can, Not As I Say: Grounding Language in Robotic Affordances",
|
||
url: "https://arxiv.org/abs/2204.01691",
|
||
topics: ["Agent"],
|
||
contribution: "SayCan 将 LLM 给出的任务相关性与技能 value/affordance 相乘,过滤语言合理但世界中不可行的动作。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "MRKL Systems: A modular, neuro-symbolic architecture that combines large language models, external knowledge sources and discrete reasoning",
|
||
url: "https://arxiv.org/abs/2205.00445",
|
||
topics: ["Agent"],
|
||
contribution: "让 LLM 作为路由器组合计算器、知识库与离散推理模块,建立模块化工具系统主线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "PAL: Program-aided Language Models",
|
||
url: "https://arxiv.org/abs/2211.10435",
|
||
topics: ["推理", "Agent"],
|
||
contribution: "让语言模型生成程序作为中间表示,再由解释器承担精确执行。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Code as Policies: Language Model Programs for Embodied Control",
|
||
url: "https://arxiv.org/abs/2209.07753",
|
||
topics: ["Agent"],
|
||
contribution: "用可组合代码表达机器人策略,把自然语言任务映射到可执行控制程序。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Generative Agents: Interactive Simulacra of Human Behavior",
|
||
url: "https://arxiv.org/abs/2304.03442",
|
||
topics: ["Agent"],
|
||
contribution: "用记忆流、反思与计划驱动长期社会行为,建立 Agent 记忆系统的经典结构。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "ReWOO: Decoupling Reasoning from Observations for Efficient Augmented Language Models",
|
||
url: "https://arxiv.org/abs/2305.18323",
|
||
topics: ["Agent", "推理"],
|
||
contribution: "先由 Planner 写含变量依赖的计划,再由 Worker 执行工具,减少重复 observation 与模型调用。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs",
|
||
url: "https://arxiv.org/abs/2307.16789",
|
||
topics: ["Agent", "数据"],
|
||
contribution: "从 16,464 个真实 REST APIs 合成指令与调用路径,并训练检索增强的 ToolLLaMA。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "HuggingGPT: Solving AI Tasks with ChatGPT and its Friends in Hugging Face",
|
||
url: "https://arxiv.org/abs/2303.17580",
|
||
topics: ["Agent", "多模态"],
|
||
contribution: "以语言模型为控制器,规划并调度模型工具库完成跨模态任务。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "CAMEL: Communicative Agents for Mind Exploration of Large Scale Language Model Society",
|
||
url: "https://arxiv.org/abs/2303.17760",
|
||
topics: ["Agent"],
|
||
contribution: "用 inception prompting 与角色扮演协议研究多 Agent 自主通信和任务协作。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "MetaGPT: Meta Programming for A Multi-Agent Collaborative Framework",
|
||
url: "https://arxiv.org/abs/2308.00352",
|
||
topics: ["Agent"],
|
||
contribution: "把软件工程 SOP、角色和中间文档写入多 Agent 协作流程。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "AutoGen: Enabling Next-Gen LLM Applications via Multi-Agent Conversation",
|
||
url: "https://arxiv.org/abs/2308.08155",
|
||
topics: ["Agent"],
|
||
contribution: "用可对话 Agent 抽象组合模型、工具、人类与多种交互拓扑。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Language Agent Tree Search Unifies Reasoning, Acting, and Planning in Language Models",
|
||
url: "https://arxiv.org/abs/2310.04406",
|
||
topics: ["Agent", "推理"],
|
||
contribution: "LATS 用树搜索、环境反馈和 value judgment 统一 reasoning、acting 与 planning。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "MemGPT: Towards LLMs as Operating Systems",
|
||
url: "https://arxiv.org/abs/2310.08560",
|
||
topics: ["Agent", "长上下文"],
|
||
contribution: "用 main、recall 与 archival memory 分层管理有效上下文,类比操作系统虚拟内存。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Mind2Web: Towards a Generalist Agent for the Web",
|
||
url: "https://arxiv.org/abs/2306.06070",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "以真实网站轨迹研究 element selection、action prediction 与跨任务/站点/领域泛化。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "WebArena: A Realistic Web Environment for Building Autonomous Agents",
|
||
url: "https://arxiv.org/abs/2307.13854",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "提供可自托管、功能完整的多站点环境,并以数据库和页面状态检查任务结果。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "GAIA: A Benchmark for General AI Assistants",
|
||
url: "https://arxiv.org/abs/2311.12983",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "用 466 个真实助手问题联合考察推理、检索、工具与多步执行。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "VisualWebArena: Evaluating Multimodal Agents on Realistic Visually Grounded Web Tasks",
|
||
url: "https://arxiv.org/abs/2401.13649",
|
||
topics: ["Agent", "多模态", "评测"],
|
||
contribution: "把视觉 grounding、图片与布局信息加入可执行的真实网页任务。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "AgentDojo: A Dynamic Environment to Evaluate Prompt Injection Attacks and Defenses for LLM Agents",
|
||
url: "https://arxiv.org/abs/2406.13352",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "以 97 个任务和 629 个安全测试,在动态工具环境中同时测 utility 与间接提示注入。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Agentless: Demystifying LLM-based Software Engineering Agents",
|
||
url: "https://arxiv.org/abs/2407.01489",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "以定位、修复、验证三阶段简化流程,证明复杂自治 harness 并非默认更优。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "ToolSandbox: A Stateful, Conversational, Interactive Evaluation Benchmark for LLM Tool Use Capabilities",
|
||
url: "https://arxiv.org/abs/2408.04682",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "引入隐式状态依赖、on-policy 用户模拟、milestone、minefield 与动态轨迹评测。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Agent Security Bench: Formalizing and Benchmarking Attacks and Defenses in LLM-based Agents",
|
||
url: "https://arxiv.org/abs/2410.02644",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "跨 system/user prompt、工具与记忆形式化 Agent 攻击面,并联合评估效用与安全。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Training Software Engineering Agents and Verifiers with SWE-Gym",
|
||
url: "https://arxiv.org/abs/2412.21139",
|
||
topics: ["Agent", "后训练", "评测"],
|
||
contribution: "构建 2,438 个可执行 Python 软件任务,用于训练软件 Agent 与 verifier。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "The Berkeley Function Calling Leaderboard: From Tool Use to Agentic Evaluation of Large Language Models",
|
||
url: "https://proceedings.mlr.press/v267/patil25a.html",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "以 AST 与执行检查覆盖串行、并行、abstention 和多轮 function calling。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "τ²-bench: Evaluating Conversational Agents in a Dual-Control Environment",
|
||
url: "https://arxiv.org/abs/2506.07982",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "用 Dec-POMDP 建模 Agent 与用户双方都能调用工具、共同改变环境的双控制任务。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning",
|
||
url: "https://arxiv.org/abs/2504.20073",
|
||
topics: ["Agent", "后训练", "推理"],
|
||
contribution: "提出 StarPO,并用 Echo Trap 揭示多轮 Agent RL 的 reward cliff、梯度 spike 与 collapse。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Agent Lightning: Train ANY AI Agents with Reinforcement Learning",
|
||
url: "https://arxiv.org/abs/2508.03680",
|
||
topics: ["Agent", "后训练", "训练系统"],
|
||
contribution: "以统一 transition 接口、分层 credit assignment 和训练—Agent 解耦连接任意运行时与 RL。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Training Long-Context, Multi-Turn Software Engineering Agents with Reinforcement Learning",
|
||
url: "https://arxiv.org/abs/2508.03501",
|
||
topics: ["Agent", "后训练", "长上下文", "训练系统"],
|
||
contribution: "在 131K context 的可执行 SWE 环境中研究终局测试奖励、长轨迹与多轮 RL。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "MCPMark: A Benchmark for Stress-Testing Realistic and Comprehensive MCP Use",
|
||
url: "https://arxiv.org/abs/2509.24002",
|
||
topics: ["Agent", "评测"],
|
||
contribution: "在真实 MCP 工具协议与服务器生态中压力测试选择、组合和长程执行。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2026,
|
||
title: "AgentENV",
|
||
url: "https://github.com/kvcache-ai/AgentENV",
|
||
topics: ["Agent", "训练系统"],
|
||
contribution: "面向 Agentic AI 的 Firecracker microVM 环境,支持 pause、resume、fork、snapshot 与高密度运行。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2012,
|
||
title: "ImageNet Classification with Deep Convolutional Neural Networks",
|
||
url: "https://proceedings.neurips.cc/paper/4824-imagenet-classification-with-deep-convolutional-neural-networks",
|
||
topics: ["多模态"],
|
||
contribution: "AlexNet 以 GPU、大规模卷积网络与 ImageNet 监督推动可扩展视觉表示。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2015,
|
||
title: "Deep Residual Learning for Image Recognition",
|
||
url: "https://arxiv.org/abs/1512.03385",
|
||
topics: ["多模态", "Transformer"],
|
||
contribution: "残差连接让更深视觉骨干稳定优化,成为多模态视觉塔的重要前史。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale",
|
||
url: "https://arxiv.org/abs/2010.11929",
|
||
topics: ["多模态"],
|
||
contribution: "ViT 把图像 patch 视为 Token 输入 Transformer。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "Scaling Up Visual and Vision-Language Representation Learning With Noisy Text Supervision",
|
||
url: "https://arxiv.org/abs/2102.05918",
|
||
topics: ["多模态", "数据", "Scaling"],
|
||
contribution: "ALIGN 用超大规模噪声 web 图文对训练双编码器,展示数据规模路线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "Learning Transferable Visual Models From Natural Language Supervision",
|
||
url: "https://arxiv.org/abs/2103.00020",
|
||
topics: ["多模态"],
|
||
contribution: "CLIP 以海量图文对比学习建立可迁移视觉语义空间。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "Multimodal Few-Shot Learning with Frozen Language Models",
|
||
url: "https://arxiv.org/abs/2106.13884",
|
||
topics: ["多模态"],
|
||
contribution: "Frozen 用视觉编码器生成连续前缀,连接冻结语言模型并激活多模态 in-context learning。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2021,
|
||
title: "LiT: Zero-Shot Transfer with Locked-image Text Tuning",
|
||
url: "https://arxiv.org/abs/2111.07991",
|
||
topics: ["多模态", "后训练"],
|
||
contribution: "锁定强图像塔,只训练文本侧去匹配既有视觉空间。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation",
|
||
url: "https://arxiv.org/abs/2201.12086",
|
||
topics: ["多模态", "数据"],
|
||
contribution: "用 captioner 生成描述、filter 清理噪声,同时服务视觉理解与生成。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Flamingo: a Visual Language Model for Few-Shot Learning",
|
||
url: "https://arxiv.org/abs/2204.14198",
|
||
topics: ["多模态"],
|
||
contribution: "用 Perceiver Resampler 与 cross-attention 连接冻结视觉/语言模型。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models",
|
||
url: "https://arxiv.org/abs/2301.12597",
|
||
topics: ["多模态"],
|
||
contribution: "用轻量 Q-Former 桥接冻结视觉编码器与 LLM。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "MiniGPT-4: Enhancing Vision-language Understanding with Advanced Large Language Models",
|
||
url: "https://arxiv.org/abs/2304.10592",
|
||
topics: ["多模态", "后训练"],
|
||
contribution: "用少量高质量视觉对话对齐冻结视觉编码器与冻结语言模型。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning",
|
||
url: "https://arxiv.org/abs/2305.06500",
|
||
topics: ["多模态", "后训练"],
|
||
contribution: "让 Q-Former 感知指令,并系统研究跨任务视觉指令泛化。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "PaLI-X: On Scaling up a Multilingual Vision and Language Model",
|
||
url: "https://arxiv.org/abs/2305.18565",
|
||
topics: ["多模态", "Scaling"],
|
||
contribution: "从组件规模与多语言任务混合两轴扩展视觉—语言模型。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "OCRBench: On the Hidden Mystery of OCR in Large Multimodal Models",
|
||
url: "https://arxiv.org/abs/2305.07895",
|
||
topics: ["多模态", "评测"],
|
||
contribution: "联合评测文字识别、场景文字 VQA、文档 VQA、KIE 与手写公式。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Evaluating Object Hallucination in Large Vision-Language Models",
|
||
url: "https://arxiv.org/abs/2305.10355",
|
||
topics: ["多模态", "评测"],
|
||
contribution: "POPE 用轮询式对象存在性问题更稳定地诊断视觉幻觉。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Kosmos-2: Grounding Multimodal Large Language Models to the World",
|
||
url: "https://arxiv.org/abs/2306.14824",
|
||
topics: ["多模态"],
|
||
contribution: "用 location token 和 grounded image-text pairs 把文本指称绑定到视觉区域。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Patch n' Pack: NaViT, a Vision Transformer for any Aspect Ratio and Resolution",
|
||
url: "https://arxiv.org/abs/2307.06304",
|
||
topics: ["多模态", "训练系统"],
|
||
contribution: "在固定 Token 预算内打包不同分辨率与长宽比图像的 patch 序列。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "MMBench: Is Your Multi-modal Model an All-around Player?",
|
||
url: "https://arxiv.org/abs/2307.06281",
|
||
topics: ["多模态", "评测"],
|
||
contribution: "用双语多选、能力分层与 CircularEval 建立综合 VLM 评测。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond",
|
||
url: "https://arxiv.org/abs/2308.12966",
|
||
topics: ["多模态"],
|
||
contribution: "把开放视觉对话、OCR 与 grounding 能力整合进 Qwen 视觉语言路线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Improved Baselines with Visual Instruction Tuning",
|
||
url: "https://arxiv.org/abs/2310.03744",
|
||
topics: ["多模态", "后训练"],
|
||
contribution: "LLaVA-1.5 用两层 MLP、更多任务数据与清晰训练配方强化简单基线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "MathVista: Evaluating Mathematical Reasoning of Foundation Models in Visual Contexts",
|
||
url: "https://arxiv.org/abs/2310.02255",
|
||
topics: ["多模态", "推理", "评测"],
|
||
contribution: "用图表、几何、函数与视觉数学题联合考察感知和多步推理。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "HallusionBench: An Advanced Diagnostic Suite for Entangled Language Hallucination and Visual Illusion",
|
||
url: "https://arxiv.org/abs/2310.14566",
|
||
topics: ["多模态", "评测"],
|
||
contribution: "用控制组问题区分语言幻觉、视觉错觉、回答倾向与逻辑一致性。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Video-LLaVA: Learning United Visual Representation by Alignment Before Projection",
|
||
url: "https://arxiv.org/abs/2311.10122",
|
||
topics: ["多模态"],
|
||
contribution: "在投影前对齐图像和视频表示,探索共同视觉—语言空间。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI",
|
||
url: "https://arxiv.org/abs/2311.16502",
|
||
topics: ["多模态", "推理", "评测"],
|
||
contribution: "以 11.5K 大学级问题覆盖六大学科、183 子领域与 30 种图像类型。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "LLaMA-VID: An Image is Worth 2 Tokens in Large Language Models",
|
||
url: "https://arxiv.org/abs/2311.17043",
|
||
topics: ["多模态", "长上下文"],
|
||
contribution: "用内容 Token 与上下文 Token 压缩长视频视觉序列。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "RLHF-V: Towards Trustworthy MLLMs via Behavior Alignment from Fine-grained Correctional Human Feedback",
|
||
url: "https://arxiv.org/abs/2312.00849",
|
||
topics: ["多模态", "后训练", "评测"],
|
||
contribution: "以 segment-level 人工纠错和 dense DPO 对齐视觉忠实度。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Visual Instruction Tuning",
|
||
url: "https://arxiv.org/abs/2304.08485",
|
||
topics: ["多模态", "后训练"],
|
||
contribution: "LLaVA 用合成视觉指令数据把 CLIP 接入语言模型。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Sigmoid Loss for Language Image Pre-Training",
|
||
url: "https://arxiv.org/abs/2303.15343",
|
||
topics: ["多模态"],
|
||
contribution: "SigLIP 以成对 sigmoid loss 简化大规模图文对比。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "DeepSeek-VL: Towards Real-World Vision-Language Understanding",
|
||
url: "https://arxiv.org/abs/2403.05525",
|
||
topics: ["多模态", "数据"],
|
||
contribution: "面向截图、PDF、OCR 与图表,用混合视觉编码器处理高分辨率并管理模态竞争。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "How Far Are We to GPT-4V? Closing the Gap to Commercial Multimodal Models with Open-Source Suites",
|
||
url: "https://arxiv.org/abs/2404.16821",
|
||
topics: ["多模态", "Scaling"],
|
||
contribution: "InternVL 1.5 结合 6B 视觉塔、最多 40 个动态 tile 与高质量双语数据。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Chameleon: Mixed-Modal Early-Fusion Foundation Models",
|
||
url: "https://arxiv.org/abs/2405.09818",
|
||
topics: ["多模态", "Transformer"],
|
||
contribution: "把图像与文字离散成可交错 Token,用 early-fusion 自回归主干统一建模。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Video-MME: The First-Ever Comprehensive Evaluation Benchmark of Multi-modal LLMs in Video Analysis",
|
||
url: "https://arxiv.org/abs/2405.21075",
|
||
topics: ["多模态", "评测"],
|
||
contribution: "用 900 段短中长视频、字幕和音频条件系统评估视频理解。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Visual Sketchpad: Sketching as a Visual Chain of Thought for Multimodal Language Models",
|
||
url: "https://arxiv.org/abs/2406.09403",
|
||
topics: ["多模态", "推理", "Agent"],
|
||
contribution: "让模型画线、框选并调用视觉工具,以新视觉产物继续推理。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "SOLO: A Single Transformer for Scalable Vision-Language Modeling",
|
||
url: "https://arxiv.org/abs/2407.06438",
|
||
topics: ["多模态", "Transformer", "Scaling"],
|
||
contribution: "探索单一 Transformer 视觉—语言架构与可稳定训练的开放配方。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "LLaVA-OneVision: Easy Visual Task Transfer",
|
||
url: "https://arxiv.org/abs/2408.03326",
|
||
topics: ["多模态"],
|
||
contribution: "统一单图、多图与视频的视觉指令迁移。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Transfusion: Predict the Next Token and Diffuse Images with One Multi-Modal Model",
|
||
url: "https://arxiv.org/abs/2408.11039",
|
||
topics: ["多模态", "Transformer"],
|
||
contribution: "在同一 Transformer 上对文字做 NTP、对连续图像表示做 diffusion。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution",
|
||
url: "https://arxiv.org/abs/2409.12191",
|
||
topics: ["多模态", "长上下文"],
|
||
contribution: "引入动态视觉 Token、M-RoPE 与统一图像 / 视频输入。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Janus: Decoupling Visual Encoding for Unified Multimodal Understanding and Generation",
|
||
url: "https://arxiv.org/abs/2410.13848",
|
||
topics: ["多模态"],
|
||
contribution: "解耦理解与生成的视觉编码路径,同时共享一个自回归主干。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "JanusFlow: Harmonizing Autoregression and Rectified Flow for Unified Multimodal Understanding and Generation",
|
||
url: "https://arxiv.org/abs/2411.07975",
|
||
topics: ["多模态"],
|
||
contribution: "把语言自回归与图像 rectified flow 放进同一框架,并对齐解耦视觉表示。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Expanding Performance Boundaries of Open-Source Multimodal Models with Model, Data, and Test-Time Scaling",
|
||
url: "https://arxiv.org/abs/2412.05271",
|
||
topics: ["多模态", "Scaling", "推理"],
|
||
contribution: "InternVL 2.5 从模型、数据与测试时配置三个轴研究多模态扩展。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "DeepSeek-VL2: Mixture-of-Experts Vision-Language Models for Advanced Multimodal Understanding",
|
||
url: "https://arxiv.org/abs/2412.10302",
|
||
topics: ["多模态", "MoE", "长上下文"],
|
||
contribution: "用动态 tile、2×2 pixel shuffle、MLP adaptor、DeepSeekMoE 与 MLA 处理高分辨率输入。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling",
|
||
url: "https://arxiv.org/abs/2501.17811",
|
||
topics: ["多模态", "Scaling"],
|
||
contribution: "延续 Janus 解耦视觉路径,改进训练策略、数据规模与模型规模。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Qwen2.5-VL Technical Report",
|
||
url: "https://arxiv.org/abs/2502.13923",
|
||
topics: ["多模态", "Agent"],
|
||
contribution: "从头训练动态分辨率 ViT,引入绝对时间编码并强化文档和视觉 Agent。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Vision-R1: Incentivizing Reasoning Capability in Multimodal Large Language Models",
|
||
url: "https://arxiv.org/abs/2503.06749",
|
||
topics: ["多模态", "推理", "后训练"],
|
||
contribution: "探索用大规模强化学习激活多模态长思维与视觉推理。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Kimi-VL Technical Report",
|
||
url: "https://arxiv.org/abs/2504.07491",
|
||
topics: ["多模态", "MoE"],
|
||
contribution: "MoonViT 原生分辨率视觉编码器与稀疏语言主干。",
|
||
spotlight: "Kimi",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "VTool-R1: VLMs Learn to Think with Images via Reinforcement Learning on Multimodal Tool Use",
|
||
url: "https://arxiv.org/abs/2505.19255",
|
||
topics: ["多模态", "推理", "Agent", "后训练"],
|
||
contribution: "用可验证奖励训练模型自主选择视觉工具并从新图像观察继续推理。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "DeepSeek-OCR: Contexts Optical Compression",
|
||
url: "https://arxiv.org/abs/2510.18234",
|
||
topics: ["多模态", "长上下文", "数据"],
|
||
contribution: "用 DeepEncoder 将二维页面压成少量视觉 Token,研究光学长上下文压缩。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2026,
|
||
title: "DeepSeek-OCR 2: Visual Causal Flow",
|
||
url: "https://arxiv.org/abs/2601.20552",
|
||
topics: ["多模态", "长上下文"],
|
||
contribution: "DeepEncoder V2 根据图像语义动态重排视觉 Token,再交给语言解码器。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2026,
|
||
title: "ToolsRL: Reward Is All Tool Learning Needs",
|
||
url: "https://arxiv.org/abs/2604.19945",
|
||
topics: ["多模态", "推理", "Agent", "后训练"],
|
||
contribution: "以强化学习研究视觉工具选择、调用与多步观察闭环。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "Measuring Massive Multitask Language Understanding",
|
||
url: "https://arxiv.org/abs/2009.03300",
|
||
topics: ["评测"],
|
||
contribution: "MMLU 用 57 学科统一衡量广泛知识与解题能力。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Beyond the Imitation Game: Quantifying and Extrapolating the Capabilities of Language Models",
|
||
url: "https://arxiv.org/abs/2206.04615",
|
||
topics: ["评测"],
|
||
contribution: "BIG-bench 汇集大量任务研究规模与能力边界。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Holistic Evaluation of Language Models",
|
||
url: "https://arxiv.org/abs/2211.09110",
|
||
topics: ["评测"],
|
||
contribution: "HELM 强调多场景、多指标、透明配置的整体评测。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding",
|
||
url: "https://arxiv.org/abs/2308.14508",
|
||
topics: ["评测", "长上下文"],
|
||
contribution: "跨中英与多任务评估长上下文理解。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "GPQA: A Graduate-Level Google-Proof Q&A Benchmark",
|
||
url: "https://arxiv.org/abs/2311.12022",
|
||
topics: ["评测", "推理"],
|
||
contribution: "由领域专家编写、难以搜索的研究生级科学问答。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "Humanity's Last Exam",
|
||
url: "https://arxiv.org/abs/2501.14249",
|
||
topics: ["评测", "推理"],
|
||
contribution: "覆盖多学科的高难、抗饱和闭答与多模态题集。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "DeepSeek-V3.2: Pushing the Frontier of Open Large Language Models",
|
||
url: "https://arxiv.org/abs/2512.02556",
|
||
topics: ["长上下文", "推理", "Agent"],
|
||
contribution: "DeepSeek Sparse Attention、规模化 RL 与 Agentic 任务合成。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2026,
|
||
title: "DeepSeek-V4: Towards Highly Efficient Million-Token Context Intelligence",
|
||
url: "https://arxiv.org/abs/2606.19348",
|
||
topics: ["基础", "长上下文", "MoE", "数据", "训练系统", "推理", "Agent"],
|
||
contribution: "32/33T 长文与 agentic 数据、CSA/HCA、mHC、Muon 和 1M 上下文双模型。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2026,
|
||
title: "Kimi K3: Open Frontier Intelligence",
|
||
url: "https://arxiv.org/abs/2607.24653",
|
||
topics: ["基础", "长上下文", "MoE", "数据", "训练系统", "低精度", "推理", "Agent", "多模态", "评测"],
|
||
contribution: "四文本域与视觉语料、2.8T-A104B、KDA/MLA、Stable LatentMoE 与 1M Agentic RL。",
|
||
spotlight: "Kimi",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2020,
|
||
title: "Clockwork: Predictable and Efficient Deep Learning Inference",
|
||
url: "https://www.usenix.org/conference/osdi20/presentation/gujarati",
|
||
topics: ["推理服务", "训练系统"],
|
||
contribution: "用可预测执行、集中式调度和显式模型缓存控制 DNN 服务尾延迟。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "ZeRO-Inference: Democratizing Massive Model Inference",
|
||
url: "https://arxiv.org/abs/2207.00032",
|
||
topics: ["推理服务", "训练系统", "低精度"],
|
||
contribution: "把巨量权重卸载到 CPU / NVMe,并通过分层预取和并行减少 GPU 容量门槛。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Orca: A Distributed Serving System for Transformer-Based Generative Models",
|
||
url: "https://www.usenix.org/conference/osdi22/presentation/yu",
|
||
topics: ["推理服务", "训练系统"],
|
||
contribution: "以 iteration-level scheduling 和 selective batching 奠定连续批处理主线。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Petals: Collaborative Inference and Fine-tuning of Large Models",
|
||
url: "https://arxiv.org/abs/2209.01188",
|
||
topics: ["推理服务", "训练系统"],
|
||
contribution: "让参与者跨互联网协作托管模型层,暴露异构、不可靠网络下的服务边界。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2022,
|
||
title: "Fast Inference from Transformers via Speculative Decoding",
|
||
url: "https://arxiv.org/abs/2211.17192",
|
||
topics: ["推理服务", "推理"],
|
||
contribution: "用便宜 draft model 提议多个 Token,再由目标模型并行验收且保持目标分布。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Accelerating Large Language Model Decoding with Speculative Sampling",
|
||
url: "https://arxiv.org/abs/2302.01318",
|
||
topics: ["推理服务", "推理"],
|
||
contribution: "独立建立保分布的推测采样算法,减少昂贵模型的串行调用轮数。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "FlexGen: High-Throughput Generative Inference of Large Language Models with a Single GPU",
|
||
url: "https://arxiv.org/abs/2303.06865",
|
||
topics: ["推理服务", "训练系统", "低精度"],
|
||
contribution: "联合 GPU、CPU 与磁盘 offload,并以搜索策略优化吞吐导向的单卡大模型推理。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "AlpaServe: Statistical Multiplexing with Model Parallelism for Deep Learning Serving",
|
||
url: "https://arxiv.org/abs/2302.11665",
|
||
topics: ["推理服务", "训练系统"],
|
||
contribution: "把统计复用、模型并行和模型放置结合,用于突发式大模型工作负载。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "FastServe: Fast Distributed Inference Serving for Large Language Models",
|
||
url: "https://arxiv.org/abs/2305.05920",
|
||
topics: ["推理服务", "训练系统"],
|
||
contribution: "借鉴多级反馈队列和抢占,降低自回归请求的 head-of-line blocking 与尾延迟。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "SpecInfer: Accelerating Generative Large Language Model Serving with Tree-based Speculative Inference and Verification",
|
||
url: "https://arxiv.org/abs/2305.09781",
|
||
topics: ["推理服务", "推理"],
|
||
contribution: "用树状候选和并行验证扩展推测解码,使一次目标调用覆盖多条草稿分支。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Efficient Memory Management for Large Language Model Serving with PagedAttention",
|
||
url: "https://arxiv.org/abs/2309.06180",
|
||
topics: ["推理服务", "长上下文", "训练系统"],
|
||
contribution: "vLLM 把 KV Cache 分页,支持非连续分配、共享与 copy-on-write。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "CacheGen: KV Cache Compression and Streaming for Fast Large Language Model Serving",
|
||
url: "https://arxiv.org/abs/2310.07240",
|
||
topics: ["推理服务", "长上下文", "低精度"],
|
||
contribution: "压缩并流式传输 KV Cache,以降低远端上下文加载延迟。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "SGLang: Efficient Execution of Structured Language Model Programs",
|
||
url: "https://arxiv.org/abs/2312.07104",
|
||
topics: ["推理服务", "Agent", "长上下文"],
|
||
contribution: "用 RadixAttention 自动复用共享前缀,并为结构化生成提供语言与运行时。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Splitwise: Efficient Generative LLM Inference Using Phase Splitting",
|
||
url: "https://arxiv.org/abs/2311.18677",
|
||
topics: ["推理服务", "训练系统"],
|
||
contribution: "将 prompt processing 与 token generation 拆到适配的硬件池并独立调度。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "Prompt Cache: Modular Attention Reuse for Low-Latency Inference",
|
||
url: "https://arxiv.org/abs/2311.04934",
|
||
topics: ["推理服务", "长上下文"],
|
||
contribution: "用模块化 prompt schema 标注可复用片段并缓存中间注意力状态。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving",
|
||
url: "https://arxiv.org/abs/2401.09670",
|
||
topics: ["推理服务", "训练系统"],
|
||
contribution: "以满足 TTFT / TPOT SLO 的 goodput 为目标,分离并独立扩容 prefill 与 decode。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads",
|
||
url: "https://arxiv.org/abs/2401.10774",
|
||
topics: ["推理服务", "推理"],
|
||
contribution: "在目标模型上增加多个 decoding heads,构造树状候选并减少独立 draft 成本。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "EAGLE: Speculative Sampling Requires Rethinking Feature Uncertainty",
|
||
url: "https://arxiv.org/abs/2401.15077",
|
||
topics: ["推理服务", "推理"],
|
||
contribution: "在 feature 层自回归预测并结合已知 Token,提升草稿准确率与推测解码速度。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "KVQuant: Towards 10 Million Context Length LLM Inference with KV Cache Quantization",
|
||
url: "https://arxiv.org/abs/2401.18079",
|
||
topics: ["推理服务", "低精度", "长上下文"],
|
||
contribution: "按通道量化 key、按 token 量化 value,并分离 outlier 以压缩长上下文 KV。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache",
|
||
url: "https://arxiv.org/abs/2402.02750",
|
||
topics: ["推理服务", "低精度", "长上下文"],
|
||
contribution: "利用 key / value 不同分布采用非对称 2-bit 量化,并保留近期高精度 residual。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Hydragen: High-Throughput LLM Inference with Shared Prefixes",
|
||
url: "https://arxiv.org/abs/2402.05099",
|
||
topics: ["推理服务", "长上下文"],
|
||
contribution: "将共享前缀与独有后缀注意力拆开,跨请求批量复用前缀计算。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "ChunkAttention: Efficient Self-Attention with Prefix-Aware KV Cache and Two-Phase Partition",
|
||
url: "https://arxiv.org/abs/2402.15220",
|
||
topics: ["推理服务", "长上下文"],
|
||
contribution: "以前缀树组织共享 KV chunks,并用两阶段分区提高内存与计算复用。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Sarathi-Serve: Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve",
|
||
url: "https://arxiv.org/abs/2403.02310",
|
||
topics: ["推理服务", "训练系统"],
|
||
contribution: "用 chunked prefill 与 stall-free scheduling 缓和长 prefill 对 decode 的干扰。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "QServe: W4A8KV4 Quantization and System Co-design for Efficient LLM Serving",
|
||
url: "https://arxiv.org/abs/2405.04532",
|
||
topics: ["推理服务", "低精度"],
|
||
contribution: "联合 W4A8KV4、数据布局和硬件友好 kernel 优化端到端服务。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "CacheBlend: Fast Large Language Model Serving for RAG with Cached Knowledge Fusion",
|
||
url: "https://arxiv.org/abs/2405.16444",
|
||
topics: ["推理服务", "长上下文"],
|
||
contribution: "融合独立缓存的文本块,并选择性重算跨块依赖以接近完整 prefill。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Llumnix: Dynamic Scheduling for Large Language Model Serving",
|
||
url: "https://arxiv.org/abs/2406.03243",
|
||
topics: ["推理服务", "训练系统"],
|
||
contribution: "用跨实例实时迁移重平衡请求、处理碎片并支持主动容错。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "MemServe: Context Caching for Disaggregated LLM Serving with Elastic Memory Pool",
|
||
url: "https://arxiv.org/abs/2406.17565",
|
||
topics: ["推理服务", "长上下文", "训练系统"],
|
||
contribution: "把独立 KV 缓存池连接到弹性 prefill / decode 服务,形成存算分离架构。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Preble: Efficient Distributed Prompt Scheduling for LLM Serving",
|
||
url: "https://arxiv.org/abs/2407.00023",
|
||
topics: ["推理服务", "长上下文", "训练系统"],
|
||
contribution: "在分布式 prompt 服务中联合前缀复用、实例负载与路由决策。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "FlashAttention-3: Fast and Accurate Attention with Asynchrony and Low-precision",
|
||
url: "https://arxiv.org/abs/2407.08608",
|
||
topics: ["推理服务", "低精度", "长上下文"],
|
||
contribution: "面向 Hopper 用异步、warp specialization 与低精度流水进一步优化精确注意力。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "EAGLE-2: Faster Inference of Language Models with Dynamic Draft Trees",
|
||
url: "https://arxiv.org/abs/2406.16858",
|
||
topics: ["推理服务", "推理"],
|
||
contribution: "根据上下文置信度动态调整草稿树,而不是固定候选拓扑。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "FlashInfer: Efficient and Customizable Attention Engine for LLM Inference Serving",
|
||
url: "https://arxiv.org/abs/2501.01005",
|
||
topics: ["推理服务", "长上下文"],
|
||
contribution: "提供面向多种 KV 布局、batch 形状与生成阶段的可组合 attention / sampling kernel。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "EAGLE-3: Scaling up Inference Acceleration of Large Language Models via Training-Time Test",
|
||
url: "https://arxiv.org/abs/2503.01840",
|
||
topics: ["推理服务", "推理"],
|
||
contribution: "直接预测 Token,并融合低、中、高层特征构造更强的推测草稿。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "DeepGEMM",
|
||
url: "https://github.com/deepseek-ai/DeepGEMM",
|
||
topics: ["推理服务", "低精度", "训练系统"],
|
||
contribution: "DeepSeek 官方开源 FP8 与低精度 GEMM kernel 库,覆盖 dense 与 MoE 形状。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2025,
|
||
title: "FlashMLA",
|
||
url: "https://github.com/deepseek-ai/FlashMLA",
|
||
topics: ["推理服务", "长上下文", "训练系统"],
|
||
contribution: "DeepSeek 官方面向 Multi-head Latent Attention prefill / decode 的高效 kernel。",
|
||
spotlight: "DeepSeek",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "REST: Retrieval-Based Speculative Decoding",
|
||
url: "https://arxiv.org/abs/2311.08252",
|
||
topics: ["推理服务", "推理"],
|
||
contribution: "从历史语料检索连续 Token 构造免训练草稿,再由目标模型并行验证。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Lookahead Decoding: Parallel Decoding without a Draft Model or Data Store",
|
||
url: "https://arxiv.org/abs/2402.02057",
|
||
topics: ["推理服务", "推理"],
|
||
contribution: "用 Jacobi 迭代并行发现并验证 n-gram,无需独立草稿模型。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "LayerSkip: Enabling Early Exit Inference and Self-Speculative Decoding",
|
||
url: "https://arxiv.org/abs/2404.16710",
|
||
topics: ["推理服务", "推理"],
|
||
contribution: "训练早退层作为同一模型的草稿,再由剩余层验证,形成自推测解码。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "H2O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models",
|
||
url: "https://arxiv.org/abs/2306.14048",
|
||
topics: ["推理服务", "长上下文"],
|
||
contribution: "识别并保留 attention heavy hitters,在有限预算下淘汰其余 KV。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "StreamingLLM: Efficient Streaming Language Models with Attention Sinks",
|
||
url: "https://arxiv.org/abs/2309.17453",
|
||
topics: ["推理服务", "长上下文"],
|
||
contribution: "保留初始 attention sinks 与近期窗口,使有限缓存支持持续流式文本。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "SnapKV: LLM Knows What You are Looking for Before Generation",
|
||
url: "https://arxiv.org/abs/2404.14469",
|
||
topics: ["推理服务", "长上下文"],
|
||
contribution: "用 prompt 末端 observation window 选择各头重要位置,压缩长 prompt KV。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "PyramidKV: Dynamic KV Cache Compression based on Pyramidal Information Funneling",
|
||
url: "https://arxiv.org/abs/2406.02069",
|
||
topics: ["推理服务", "长上下文"],
|
||
contribution: "依据层间信息聚合现象,为不同层分配递减的 KV 容量预算。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "InfiniGen: Efficient Generative Inference of Large Language Models with Dynamic KV Cache Management",
|
||
url: "https://arxiv.org/abs/2406.19707",
|
||
topics: ["推理服务", "长上下文"],
|
||
contribution: "提前预测关键 KV,并从 CPU 内存选择性加载以减少 GPU 状态容量。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "Quest: Query-Aware Sparsity for Efficient Long-Context LLM Inference",
|
||
url: "https://arxiv.org/abs/2406.10774",
|
||
topics: ["推理服务", "长上下文"],
|
||
contribution: "利用 page 级统计按 query 选择相关 KV pages,减少长上下文 decode 读取。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2024,
|
||
title: "BurstGPT: A Real-world Workload Dataset to Optimize LLM Serving Systems",
|
||
url: "https://arxiv.org/abs/2401.17644",
|
||
topics: ["推理服务", "训练系统", "评测"],
|
||
contribution: "公开真实服务 trace,揭示突发、长度和模型混合对容量规划的影响。",
|
||
verified: true,
|
||
},
|
||
{
|
||
year: 2023,
|
||
title: "TensorRT-LLM",
|
||
url: "https://github.com/NVIDIA/TensorRT-LLM",
|
||
topics: ["推理服务", "低精度", "训练系统"],
|
||
contribution: "NVIDIA 官方 LLM 推理运行时,整合低精度 kernel、并行、KV Cache 与动态批处理。",
|
||
verified: true,
|
||
},
|
||
];
|
||
|
||
export const paperTopics: PaperTopic[] = [
|
||
"基础",
|
||
"Transformer",
|
||
"长上下文",
|
||
"MoE",
|
||
"Scaling",
|
||
"数据",
|
||
"训练系统",
|
||
"优化器",
|
||
"低精度",
|
||
"后训练",
|
||
"推理",
|
||
"Agent",
|
||
"多模态",
|
||
"推理服务",
|
||
"评测",
|
||
];
|