-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathindex.html
More file actions
426 lines (393 loc) · 31.8 KB
/
Copy pathindex.html
File metadata and controls
426 lines (393 loc) · 31.8 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
<!DOCTYPE html>
<html lang="zh-CN">
<head>
<meta charset="UTF-8">
<script>
/* Set language before first paint: saved preference wins, else follow the browser */
(function(){
var lang = null;
try{ lang = localStorage.getItem('pr-lang'); }catch(e){}
if (lang !== 'zh' && lang !== 'en') {
lang = String(navigator.language || navigator.userLanguage || 'en').toLowerCase().indexOf('zh') === 0 ? 'zh' : 'en';
}
document.documentElement.lang = (lang === 'en') ? 'en' : 'zh-CN';
})();
</script>
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<meta name="description" content="Illustrated paper-reading notes in Chinese and English: efficient inference and benchmarks, every figure/table embedded at 300 DPI. 论文中英文双语精读合集,图文配合,全部图表高清内嵌。">
<meta name="keywords" content="paper reading, 论文精读, 论文解读, 高效推理, long context, quantization, multimodal embedding, benchmark">
<meta property="og:type" content="website">
<meta property="og:title" content="Paper Reading Collection · 论文精读合集">
<meta property="og:description" content="One illustrated reading page per paper — full Chinese deep-reads plus English editions, every figure at 300 DPI. 每篇论文一个图文配合的精读页,中英双语。">
<meta property="og:url" content="https://qqtang-code.github.io/Paper-Reading-Collection/">
<meta name="theme-color" content="#2f5bd9">
<title>Paper Reading Collection · 论文精读合集</title>
<style>
:root{
--bg:#f6f8fb; --card:#ffffff; --ink:#1b2432; --muted:#5c6675;
--accent:#2f5bd9; --accent-soft:#e8edfb; --teal:#0d8a6d; --teal-soft:#e3f4ee;
--border:#dde3ec; --th-bg:#eef1f7;
--topbar-bg:rgba(255,255,255,.85); --shadow:0 2px 14px rgba(16,30,60,.10);
}
html[data-theme="dark"]{
--bg:#0f1420; --card:#171e2d; --ink:#e8ecf4; --muted:#9aa6ba;
--accent:#6e95f5; --accent-soft:#1d2a47; --teal:#3dd6ae; --teal-soft:#0f2e26;
--border:#26314a; --th-bg:#1c2436;
--topbar-bg:rgba(15,20,32,.88); --shadow:0 2px 14px rgba(0,0,0,.45);
}
*{box-sizing:border-box}
html{scroll-behavior:smooth}
body{margin:0; background:var(--bg); color:var(--ink);
font-family:-apple-system,BlinkMacSystemFont,"Segoe UI","PingFang SC","Hiragino Sans GB","Microsoft YaHei",sans-serif;
line-height:1.85; font-size:16px; transition:background .25s ease,color .25s ease}
.wrap{max-width:1080px; margin:0 auto; padding:0 28px}
/* ===== inline i18n: show only the active language ===== */
html[lang="en"] .i18n-zh{display:none!important}
html[lang="zh-CN"] .i18n-en{display:none!important}
html[lang="zh"] .i18n-en{display:none!important}
.topbar{position:sticky; top:0; z-index:60; background:var(--topbar-bg);
backdrop-filter:blur(10px); -webkit-backdrop-filter:blur(10px);
border-bottom:1px solid var(--border)}
.topbar .inner{max-width:1080px; margin:0 auto; padding:10px 28px; display:flex; align-items:center; gap:14px}
.topbar .brand{font-weight:800; font-size:15px; color:var(--ink); text-decoration:none; white-space:nowrap}
.topbar .brand span{color:var(--accent)}
.topbar .links{margin-left:auto; display:flex; gap:14px; align-items:center}
.topbar a{font-size:13px; color:var(--muted); text-decoration:none}
.topbar a:hover{color:var(--accent)}
.theme-toggle{border:1px solid var(--border); background:var(--card); color:var(--ink);
border-radius:8px; padding:4px 10px; font-size:12.5px; cursor:pointer}
.lang-toggle{border:1px solid var(--border); background:var(--card); color:var(--ink);
border-radius:8px; padding:4px 10px; font-size:12.5px; cursor:pointer; font-weight:700}
.lang-toggle:hover{border-color:var(--accent); color:var(--accent)}
header.hero{background:linear-gradient(135deg,#1f2c52 0%,#2f5bd9 60%,#3f7de0 100%);
color:#fff; padding:64px 0 56px; margin:0}
header.hero .kicker{font-size:13px; letter-spacing:.18em; text-transform:uppercase; opacity:.85; font-weight:600}
header.hero h1{font-size:34px; line-height:1.35; margin:14px 0 10px; font-weight:800}
header.hero .subtitle{font-size:16.5px; opacity:.92; max-width:840px}
.pill{display:inline-block; background:rgba(255,255,255,.14); border:1px solid rgba(255,255,255,.28);
padding:2px 12px; border-radius:999px; font-size:12.5px; margin:14px 6px 0 0}
.stats{display:flex; flex-wrap:wrap; gap:12px; margin-top:22px}
.stats .s{background:rgba(255,255,255,.12); border:1px solid rgba(255,255,255,.22);
border-radius:10px; padding:8px 18px; font-size:13.5px}
.stats b{font-size:17px}
section{margin:48px 0}
.sec-no{font-size:13px; color:var(--accent); font-weight:700; letter-spacing:.14em; text-transform:uppercase; margin-bottom:4px}
h2.sec{font-size:24px; font-weight:800; margin:0 0 18px; padding-bottom:10px; border-bottom:3px solid var(--accent); display:inline-block}
.cards{display:grid; grid-template-columns:repeat(auto-fit,minmax(300px,1fr)); gap:20px}
.pcard{background:var(--card); border:1px solid var(--border); border-radius:14px;
padding:22px 24px; box-shadow:var(--shadow); display:flex; flex-direction:column; gap:10px;
transition:transform .15s ease, border-color .15s ease}
.pcard:hover{transform:translateY(-3px); border-color:var(--accent)}
.pcard .cat{font-size:12px; font-weight:700; letter-spacing:.06em; color:var(--teal);
background:var(--teal-soft); border-radius:999px; padding:2px 12px; align-self:flex-start}
.pcard h3{margin:0; font-size:18.5px; line-height:1.45}
.pcard h3 a{color:var(--ink); text-decoration:none}
.pcard h3 a:hover{color:var(--accent)}
.pcard .one{font-size:14px; color:var(--muted); margin:0; flex:1}
.pcard .meta{font-size:12.5px; color:var(--muted); border-top:1px dashed var(--border); padding-top:10px;
display:flex; flex-wrap:wrap; gap:6px 16px}
.pcard .meta a{color:var(--accent); text-decoration:none}
.pcard .go{margin-top:4px; font-size:14px}
.pcard .go a{display:inline-block; background:var(--accent); color:#fff; text-decoration:none;
border-radius:8px; padding:6px 18px; font-weight:700}
.pcard .go a:hover{filter:brightness(1.1)}
footer{margin-top:70px; padding:30px 0 46px; border-top:1px solid var(--border);
color:var(--muted); font-size:13.5px; text-align:center}
footer a{color:var(--accent); text-decoration:none}
.back-top{position:fixed; right:26px; bottom:30px; z-index:90; width:44px; height:44px; border-radius:50%;
background:var(--accent); color:#fff; border:none; font-size:20px; cursor:pointer;
box-shadow:var(--shadow); display:none}
.back-top.show{display:block}
@media print{.topbar,.back-top,.theme-toggle,.lang-toggle{display:none!important}}
@media (max-width:760px){header.hero h1{font-size:25px}}
</style>
</head>
<body>
<nav class="topbar">
<div class="inner">
<a class="brand" href="#top">
<span class="i18n-zh">Paper Reading<span> · 论文精读合集</span></span>
<span class="i18n-en">Paper Reading<span> Collection</span></span>
</a>
<div class="links">
<a href="#attention"><span class="i18n-zh">注意力与 KV Cache</span><span class="i18n-en">Attention & KV Cache</span></a>
<a href="#inference"><span class="i18n-zh">高效推理</span><span class="i18n-en">Efficient Inference</span></a>
<a href="#benchmark"><span class="i18n-zh">评测基准</span><span class="i18n-en">Benchmarks</span></a>
<a href="https://github.com/qqtang-code/Paper-Reading-Collection">GitHub</a>
<button class="lang-toggle" id="langToggle" title="Switch language / 切换语言" aria-label="Switch language / 切换语言"><span class="i18n-zh">EN</span><span class="i18n-en">中文</span></button>
<button class="theme-toggle" aria-label="切换深色/浅色 (Toggle dark/light)" title="切换深色/浅色 (Toggle dark/light)">🌓</button>
</div>
</div>
</nav>
<header class="hero" id="top">
<div class="wrap">
<div class="kicker">Paper Reading Collection · by qqtang-code</div>
<h1>
<span class="i18n-zh">论文精读合集</span>
<span class="i18n-en">Paper Reading Collection</span>
</h1>
<div class="subtitle">
<span class="i18n-zh">每篇论文一个"图文配合"的精读页:全部 Figure / Table 以 300 DPI 高清内嵌于对应讲解段落,配逐点解读、术语速查与编者点评。每篇均提供<strong>中文精读版</strong>与<strong>英文版</strong>(三篇旗舰论文为完整英译,其余为英文速读)。</span>
<span class="i18n-en">One illustrated reading page per paper: every figure and table embedded at 300 DPI in the paragraph that discusses it, with point-by-point commentary, a glossary and editorial notes. Every paper ships a <strong>full Chinese deep-read</strong> plus an <strong>English edition</strong> (complete translations for the three flagship papers, quick-read editions for the rest).</span>
</div>
<div>
<span class="pill"><span class="i18n-zh">图文配合</span><span class="i18n-en">Illustrated</span></span>
<span class="pill">300 DPI</span>
<span class="pill"><span class="i18n-zh">深色模式</span><span class="i18n-en">Dark mode</span></span>
<span class="pill">KaTeX</span>
<span class="pill"><span class="i18n-zh">打印友好</span><span class="i18n-en">Print-friendly</span></span>
<span class="pill"><span class="i18n-zh">中英双语</span><span class="i18n-en">中文 · English</span></span>
</div>
<div class="stats">
<div class="s"><span class="i18n-zh">论文 <b>9</b> 篇</span><span class="i18n-en"><b>9</b> papers</span></div>
<div class="s"><span class="i18n-zh">图表 <b>71</b> 图 + <b>85</b> 表</span><span class="i18n-en"><b>71</b> figures + <b>85</b> tables</span></div>
<div class="s"><span class="i18n-zh">分类 <b>3</b> 大类</span><span class="i18n-en"><b>3</b> categories</span></div>
<div class="s"><span class="i18n-zh">语言 <b>中 / EN</b></span><span class="i18n-en">Languages <b>ZH / EN</b></span></div>
</div>
</div>
</header>
<div class="wrap">
<section id="attention">
<div class="sec-no">Category 01</div>
<h2 class="sec">
<span class="i18n-zh">🎯 注意力与 KV Cache</span>
<span class="i18n-en">🎯 Attention & KV Cache</span>
</h2>
<div class="cards">
<div class="pcard">
<span class="cat"><span class="i18n-zh">稀疏注意力 · KV Cache · 长上下文</span><span class="i18n-en">Sparse Attention · KV Cache · Long Context</span></span>
<h3>
<a class="i18n-zh" href="attention-kv-cache/declarative-attention/Declarative-Attention论文精读_HTML.html">① Declarative Attention<br><span style="font-size:14px; font-weight:600; color:var(--muted)">Language Models Can Control Their Own Attention</span></a>
<a class="i18n-en" href="attention-kv-cache/declarative-attention/en.html">① Declarative Attention<br><span style="font-size:14px; font-weight:600; color:var(--muted)">Language Models Can Control Their Own Attention</span></a>
</h3>
<p class="one">
<span class="i18n-zh">让模型在思维链里用标签声明"自己要看哪里",引擎解析声明并跳过大部分 KV cache 读取。零训练零样本:注意力读取降 <b>52.0% / 31.1%</b>,精度仅降 1.27pp / 2.75pp,解码墙钟 <b>0.71× / 0.77×</b>。</span>
<span class="i18n-en">The model declares where to look with tags in its chain-of-thought; the engine parses the declarations and skips most KV-cache reads. Zero training, zero-shot: attention reads down <b>52.0% / 31.1%</b>, accuracy down only 1.27pp / 2.75pp, decode wall-time <b>0.71× / 0.77×</b>.</span>
</p>
<div class="meta">
<span>2026-09</span><span><span class="i18n-zh">10 图 10 表</span><span class="i18n-en">10 figures · 10 tables</span></span><span>arXiv:<a href="https://arxiv.org/abs/2609.02737">2609.02737</a></span>
<span class="i18n-zh">中文精读 + 英文完整版</span><span class="i18n-en">ZH deep-read + full EN mirror</span>
</div>
<div class="go">
<a class="i18n-zh" href="attention-kv-cache/declarative-attention/Declarative-Attention论文精读_HTML.html">开始阅读 →</a>
<a class="i18n-en" href="attention-kv-cache/declarative-attention/en.html">Read in English →</a>
</div>
</div>
<div class="pcard">
<span class="cat"><span class="i18n-zh">KV Cache 驱逐 · 推理模型 · 高效服务</span><span class="i18n-en">KV Cache Eviction · Reasoning Models · Efficient Serving</span></span>
<h3>
<a class="i18n-zh" href="attention-kv-cache/random-attention/">④ Random Attention<br><span style="font-size:14px; font-weight:600; color:var(--muted)">重新思考面向高效推理的 KV Cache 驱逐</span></a>
<a class="i18n-en" href="attention-kv-cache/random-attention/en.html">④ Random Attention<br><span style="font-size:14px; font-weight:600; color:var(--muted)">Rethinking KV Cache Eviction for Efficient Reasoning</span></a>
</h3>
<p class="one">
<span class="i18n-zh">KV cache 驱逐不需要打分:保住 prompt + 每个 KV 头内均匀随机驱逐,4 模型 × 6 任务追平最强基线、<b>60 格中 31 格显著领先</b>,vLLM 吞吐再快 <b>32–43%</b>——一篇把"选择信号几乎不贡献精度"钉死的机制论文。</span>
<span class="i18n-en">KV cache eviction needs no score: pin the prompt and evict uniformly at random within each head — matching the strongest evictor across 4 models × 6 reasoning tasks (<b>significantly ahead in 31 of 60 cells</b>) while serving <b>32–43% higher throughput</b> in vLLM, with a full mechanistic account of why the selection signal buys almost nothing.</span>
</p>
<div class="meta">
<span>2026-09</span><span><span class="i18n-zh">5 图 11 表</span><span class="i18n-en">5 figures · 11 tables</span></span><span>arXiv:<a href="https://arxiv.org/abs/2609.03430">2609.03430</a></span>
<span class="i18n-zh">中文精读 + 英文速读</span><span class="i18n-en">ZH deep-read + EN quick-read</span>
</div>
<div class="go">
<a class="i18n-zh" href="attention-kv-cache/random-attention/">开始阅读 →</a>
<a class="i18n-en" href="attention-kv-cache/random-attention/en.html">Read in English →</a>
</div>
</div>
<div class="pcard">
<span class="cat"><span class="i18n-zh">混合注意力 · 测试时自适应 · 长上下文</span><span class="i18n-en">Hybrid Attention · Test-time Adaptation · Long Context</span></span>
<h3>
<a class="i18n-zh" href="attention-kv-cache/elastic-attention/">⑦ Elastic Attention<br><span style="font-size:14px; font-weight:600; color:var(--muted)">让模型的稀疏比随输入"伸缩"</span></a>
<a class="i18n-en" href="attention-kv-cache/elastic-attention/en.html">⑦ Elastic Attention<br><span style="font-size:14px; font-weight:600; color:var(--muted)">Test-time Adaptive Sparsity Ratios for Efficient Transformers</span></a>
</h3>
<p class="one">
<span class="i18n-zh">混合注意力的 FA/SA 比例从静态超参数变成<b>输入自适应</b>:每层一个 0.27M 参数的 Attention Router,推理时把每个 KV 头分到全注意力或稀疏注意力——摘要/代码任务稀疏到 ~0.82,问答任务守住 0.63–0.68。8×A800 训练 <b>12 小时</b>、骨干完全冻结,3 个骨干在 LongBench-E 平均分全部第一,RULER 256K 外推优势最大;路由器延迟 <b>0.196 ms</b> 且与序列长度无关。</span>
<span class="i18n-en">The FA/SA ratio stops being a static hyper-parameter and becomes <b>input-adaptive</b>: a 0.27M-parameter Attention Router per layer assigns each KV head to full or sparse attention at inference — sparsity ~0.82 for summarization/code, 0.63–0.68 for QA. Trained in <b>12 hours on 8×A800</b> with the backbone frozen, it tops LongBench-E on all three backbones and leads most on RULER at 256K, with router latency of <b>0.196 ms</b> independent of sequence length.</span>
</p>
<div class="meta">
<span>2026-09</span><span><span class="i18n-zh">17 图 14 表</span><span class="i18n-en">17 figures · 14 tables</span></span><span><a href="https://github.com/qqtang-code/Elastic-Attention-Project-Page">GitHub</a> · <a href="https://openreview.net/forum?id=rLO2NTUHSW">OpenReview</a></span>
<span class="i18n-zh">中文精读 + 英文速读</span><span class="i18n-en">ZH deep-read + EN quick-read</span>
</div>
<div class="go">
<a class="i18n-zh" href="attention-kv-cache/elastic-attention/">开始阅读 →</a>
<a class="i18n-en" href="attention-kv-cache/elastic-attention/en.html">Read in English →</a>
</div>
</div>
<div class="pcard">
<span class="cat"><span class="i18n-zh">混合稀疏注意力 · 两级 KV 共享 · 长上下文 · Agentic</span><span class="i18n-en">Hybrid Sparse Attention · Two-Level KV Sharing · Long Context · Agentic</span></span>
<h3>
<a class="i18n-zh" href="attention-kv-cache/hysparse2/">⑧ HySparse2<br><span style="font-size:14px; font-weight:600; color:var(--muted)">两级 KV 共享的混合稀疏注意力</span></a>
<a class="i18n-en" href="attention-kv-cache/hysparse2/en.html">⑧ HySparse2<br><span style="font-size:14px; font-weight:600; color:var(--muted)">Hybrid Sparse Attention with Two-Level KV Sharing</span></a>
</h3>
<p class="one">
<span class="i18n-zh">把"局部性"从一条需要自身隐状态的独立 SWA 分支,改写成"最近窗口永远入选"的固定选择——cross-decoder 从此不再依赖自己的逐层隐状态,<b>prefill 只要跑完半个模型就能退出</b>。1M 上下文 prefill FLOPs 降 <b>2.92×/5.02×</b>、KV cache 压到 <b>2.69 GB</b>,后训练检索 MRCR-v2 <b>+11.30</b>、RULER-v2 <b>+19.81</b> 个百分点。</span>
<span class="i18n-en">Rewrites locality from a branch needing its own hidden states into "the recent window is always selected", so the cross-decoder no longer depends on its own layer-by-layer hidden suffix — <b>prefill can exit after half the model</b>. At 1M context: prefill FLOPs down <b>2.92×/5.02×</b>, KV cache down to <b>2.69 GB</b>, post-training retrieval up <b>+11.30</b> MRCR-v2 and <b>+19.81</b> RULER-v2 points.</span>
</p>
<div class="meta">
<span>2026-09</span><span><span class="i18n-zh">5 图 5 表</span><span class="i18n-en">5 figures · 5 tables</span></span><span>arXiv:<a href="https://arxiv.org/abs/2609.26368">2609.26368</a></span>
<span class="i18n-zh">中文精读 + 英文完整版</span><span class="i18n-en">ZH deep-read + full EN mirror</span>
</div>
<div class="go">
<a class="i18n-zh" href="attention-kv-cache/hysparse2/">开始阅读 →</a>
<a class="i18n-en" href="attention-kv-cache/hysparse2/en.html">Read in English →</a>
</div>
</div>
<div class="pcard">
<span class="cat"><span class="i18n-zh">块稀疏注意力 · 金字塔 Top-K · 长上下文 · O(N log N)</span><span class="i18n-en">Block-Sparse Attention · Pyramid Top-K · Long Context · O(N log N)</span></span>
<h3>
<a class="i18n-zh" href="attention-kv-cache/pisa/">⑨ PISA<br><span style="font-size:14px; font-weight:600; color:var(--muted)">对数线性的块稀疏注意力选择</span></a>
<a class="i18n-en" href="attention-kv-cache/pisa/en.html">⑨ PISA<br><span style="font-size:14px; font-weight:600; color:var(--muted)">Block Sparse Attention with Log-Linear Complexity</span></a>
</h3>
<p class="one">
<span class="i18n-zh">块稀疏注意力省下了注意力计算,却没省下"挑哪些块"——为每个 query 扫全部 N/C 个键块,选择阶段仍是 O(N²/C)。PISA 把键池化成 O(log N) 层金字塔,从最粗层逐层 Top-K 展开,块分数改用子块摘要上的 LogSumExp 并用 Jensen 不等式给出夹逼论证。256K 选块延迟比 BSA 快 <b>9.95×</b>,Recall@8 <b>90.95%</b> 对 BSA 的 85.91%;RULER 平均 <b>62.80</b> 对 BSA 的 54.99。16K 以内 BSA 更快,交叉点在 16K–32K 之间。</span>
<span class="i18n-en">Block-sparse attention saves the attention compute but not the cost of choosing blocks — scanning all N/C key blocks per query keeps selection at O(N²/C). PISA pools keys into an O(log N)-level pyramid, narrows Top-K level by level from the coarsest, and scores blocks with a LogSumExp over child summaries backed by a Jensen sandwich argument. Selection latency is <b>9.95×</b> faster than BSA at 256K, Recall@8 is <b>90.95%</b> against BSA's 85.91%, and the RULER average is <b>62.80</b> against 54.99. BSA is still faster below 16K; the crossover sits between 16K and 32K.</span>
</p>
<div class="meta">
<span>2026-09</span><span><span class="i18n-zh">3 图 7 表 + 1 算法框图</span><span class="i18n-en">3 figures · 7 tables · 1 algorithm</span></span><span>arXiv:<a href="https://arxiv.org/abs/2609.31093">2609.31093</a></span>
<span class="i18n-zh">中文精读 + 英文速读版</span><span class="i18n-en">ZH deep-read + EN quick-read</span>
</div>
<div class="go">
<a class="i18n-zh" href="attention-kv-cache/pisa/">开始阅读 →</a>
<a class="i18n-en" href="attention-kv-cache/pisa/en.html">Read in English →</a>
</div>
</div>
</div>
</section>
<section id="inference">
<div class="sec-no">Category 02</div>
<h2 class="sec">
<span class="i18n-zh">⚡ 高效推理与部署</span>
<span class="i18n-en">⚡ Efficient Inference & Deployment</span>
</h2>
<div class="cards">
<div class="pcard">
<span class="cat"><span class="i18n-zh">MoE · 边缘部署 · 带宽自适应</span><span class="i18n-en">MoE · Edge Deployment · Bandwidth-Adaptive</span></span>
<h3>
<a class="i18n-zh" href="efficient-inference/freetoken/">② FreeToken<br><span style="font-size:14px; font-weight:600; color:var(--muted)">边缘原生的 MoE 服务系统</span></a>
<a class="i18n-en" href="efficient-inference/freetoken/en.html">② FreeToken<br><span style="font-size:14px; font-weight:600; color:var(--muted)">Edge-Native MoE Serving with Bandwidth-Adaptive Execution</span></a>
</h3>
<p class="one">
<span class="i18n-zh">把个人电脑变成统一的弹性推理平台:带宽自适应执行让 <b>8GB 笔记本跑 35B</b>、台式机跑 284B、单张工作站 GPU 跑 <b>753B GLM-5.2</b>,快到能支撑真实 agent 负载。</span>
<span class="i18n-en">Turns the consumer PC into one elastic inference platform: bandwidth-adaptive execution puts a <b>35B model on an 8GB laptop</b>, 284B on a desktop, and <b>753B GLM-5.2 on a single workstation GPU</b> — fast enough for real agent workloads.</span>
</p>
<div class="meta">
<span>2026-09</span><span><span class="i18n-zh">5 图 1 表</span><span class="i18n-en">5 figures · 1 table</span></span><span>arXiv:<a href="https://arxiv.org/abs/2608.16157">2608.16157</a></span>
<span class="i18n-zh">中文精读 + 英文速读</span><span class="i18n-en">ZH deep-read + EN quick-read</span>
</div>
<div class="go">
<a class="i18n-zh" href="efficient-inference/freetoken/">开始阅读 →</a>
<a class="i18n-en" href="efficient-inference/freetoken/en.html">Read in English →</a>
</div>
</div>
<div class="pcard">
<span class="cat"><span class="i18n-zh">量化 · NVFP4 · 温度缩放</span><span class="i18n-en">Quantization · NVFP4 · Temperature Scaling</span></span>
<h3>
<a class="i18n-zh" href="efficient-inference/reset/">③ ReSET<br><span style="font-size:14px; font-weight:600; color:var(--muted)">Accurate Latency-Critical NVFP4 Reasoning</span></a>
<a class="i18n-en" href="efficient-inference/reset/en.html">③ ReSET<br><span style="font-size:14px; font-weight:600; color:var(--muted)">Step-Aware Temperature Scaling for NVFP4 Reasoning</span></a>
</h3>
<p class="one">
<span class="i18n-zh">NVFP4 低精度推理在延迟敏感场景的精度救星:<b>步骤感知温度缩放</b>按解码步骤动态调整 softmax 温度,不牺牲延迟地恢复量化精度。</span>
<span class="i18n-en">An accuracy rescue for latency-critical NVFP4 inference: <b>step-aware temperature scaling</b> adapts the softmax temperature per reasoning step — recovering quantized accuracy at zero weight cost, plus a CUDA-core small-M decode kernel.</span>
</p>
<div class="meta">
<span>2026-08</span><span><span class="i18n-zh">8 图 16 表</span><span class="i18n-en">8 figures · 16 tables</span></span><span>arXiv:<a href="https://arxiv.org/abs/2606.13233">2606.13233</a></span>
<span class="i18n-zh">中文精读 + 英文速读</span><span class="i18n-en">ZH deep-read + EN quick-read</span>
</div>
<div class="go">
<a class="i18n-zh" href="efficient-inference/reset/">开始阅读 →</a>
<a class="i18n-en" href="efficient-inference/reset/en.html">Read in English →</a>
</div>
</div>
<div class="pcard">
<span class="cat"><span class="i18n-zh">KV Cache 压缩 · MoE · 多模态 · 长上下文</span><span class="i18n-en">KV Cache Compression · MoE · Multimodal · Long Context</span></span>
<h3>
<a class="i18n-zh" href="efficient-inference/deepseek-v41-flash/">⑥ DeepSeek-V4.1-Flash<br><span style="font-size:14px; font-weight:600; color:var(--muted)">552B 多模态 MoE 高效推理技术报告</span></a>
<a class="i18n-en" href="efficient-inference/deepseek-v41-flash/en.html">⑥ DeepSeek-V4.1-Flash<br><span style="font-size:14px; font-weight:600; color:var(--muted)">Technical Report · 552B Multimodal MoE</span></a>
</h3>
<p class="one">
<span class="i18n-zh">DeepSeek-V4.1-Flash 技术报告:552B 多模态 MoE 用 Causal Encoder-Decoder 让 prefill 只激活 8B;CSA2 跨层 KV 复用 + FP4 量化把全局 KV Cache 压到<b>每 token 890 字节</b>(V4-Flash 的 1/4、V1 的 1/437),SWA Bounded Replay 再把持久缓存压到上代 1/8——性能反超更大的 V4-Flash:DeepSWE v1.1 <b>74.2</b>、Codeforces <b>3471</b>。</span>
<span class="i18n-en">The DeepSeek-V4.1-Flash technical report: a 552B multimodal MoE whose Causal Encoder-Decoder activates just 8B in prefill; CSA2 cross-layer KV reuse + FP4 squeeze the global KV cache to <b>890 bytes/token</b> (1/4 of V4-Flash, 1/437 of V1) while SWA Bounded Replay shrinks the persistent cache to 1/8 of the last generation — outperforming the larger V4-Flash: DeepSWE v1.1 <b>74.2</b>, Codeforces <b>3471</b>.</span>
</p>
<div class="meta">
<span>2026-09</span><span><span class="i18n-zh">12 图 5 表</span><span class="i18n-en">12 figures · 5 tables</span></span><span><a href="https://github.com/qqtang-code/DeepSeek-V4.1-Flash-Project-Page">GitHub</a></span>
<span class="i18n-zh">中文精读 + 英文完整版</span><span class="i18n-en">ZH deep-read + full EN mirror</span>
</div>
<div class="go">
<a class="i18n-zh" href="efficient-inference/deepseek-v41-flash/">开始阅读 →</a>
<a class="i18n-en" href="efficient-inference/deepseek-v41-flash/en.html">Read in English →</a>
</div>
</div>
</div>
</section>
<section id="benchmark">
<div class="sec-no">Category 03</div>
<h2 class="sec">
<span class="i18n-zh">📏 评测基准</span>
<span class="i18n-en">📏 Benchmarks</span>
</h2>
<div class="cards">
<div class="pcard">
<span class="cat"><span class="i18n-zh">多模态 · 嵌入模型 · 长上下文</span><span class="i18n-en">Multimodal · Embeddings · Long Context</span></span>
<h3>
<a class="i18n-zh" href="benchmarks/mmlongembed/">⑤ MMLongEmbed<br><span style="font-size:14px; font-weight:600; color:var(--muted)">长上下文场景下的多模态嵌入模型基准</span></a>
<a class="i18n-en" href="benchmarks/mmlongembed/en.html">⑤ MMLongEmbed<br><span style="font-size:14px; font-weight:600; color:var(--muted)">Benchmarking Multimodal Embedding Models in Long-Context Scenarios</span></a>
</h3>
<p class="one">
<span class="i18n-zh">首个专门评测多模态嵌入模型长上下文能力的基准:<b>4 个检索任务、8 个数据集、8,460 个查询、11 个模型</b>。核心发现:现役模型靠表面特征匹配"作弊",细粒度信息保持随长度衰退——"窗口大"≠"读得懂"。</span>
<span class="i18n-en">The first benchmark dedicated to MEM long-context ability: <b>4 retrieval tasks, 8 datasets, 8,460 queries, 11 models</b>. Key finding: current models "cheat" via surface-feature matching and fine-grained retention decays with length — a wide window ≠ real comprehension.</span>
</p>
<div class="meta">
<span>2026-08</span><span><span class="i18n-zh">6 图 16 表</span><span class="i18n-en">6 figures · 16 tables</span></span><span><a href="https://github.com/AmamiSora1228/MMLongEmbed">GitHub</a></span>
<span class="i18n-zh">中文精读 + 英文速读</span><span class="i18n-en">ZH deep-read + EN quick-read</span>
</div>
<div class="go">
<a class="i18n-zh" href="benchmarks/mmlongembed/">开始阅读 →</a>
<a class="i18n-en" href="benchmarks/mmlongembed/en.html">Read in English →</a>
</div>
</div>
</div>
</section>
</div><!-- /wrap -->
<button class="back-top" id="backTop" aria-label="回到顶部 (Back to top)" title="回到顶部 (Back to top)">↑</button>
<footer>
<div class="wrap">
<span class="i18n-zh">本合集由 <a href="https://github.com/qqtang-code/paper-reading-html">paper-reading-html</a> 工作流制作,图片以 300 DPI 摘自各论文原文,解读为编者观点,版权归原作者所有。首次访问跟随浏览器语言,切换后偏好会被记住。</span>
<span class="i18n-en">Built with the <a href="https://github.com/qqtang-code/paper-reading-html">paper-reading-html</a> workflow. Images are extracted from the original papers at 300 DPI; commentary reflects the editor's views; all rights belong to the original authors. First visit follows your browser language; your switch is remembered.</span>
<br>
<a href="https://github.com/qqtang-code/Paper-Reading-Collection">GitHub</a> · <span class="i18n-zh">支持深色模式 · 打印友好</span><span class="i18n-en">Dark mode supported · print-friendly</span>
</div>
</footer>
<script>
(function(){
// ===== language: preference in localStorage['pr-lang'], first visit follows the browser =====
var TITLES = {zh:'Paper Reading Collection · 论文精读合集', en:'Paper Reading Collection · Paper Reading Notes'};
function curLang(){ return document.documentElement.lang === 'en' ? 'en' : 'zh'; }
function applyTitle(){ document.title = TITLES[curLang()]; }
applyTitle();
document.getElementById('langToggle').addEventListener('click', function(){
var next = curLang() === 'en' ? 'zh' : 'en';
document.documentElement.lang = next === 'en' ? 'en' : 'zh-CN';
try{ localStorage.setItem('pr-lang', next); }catch(e){}
applyTitle();
});
// ===== dark mode: remember preference, default follows system =====
var KEY = 'pr-theme';
function apply(t){
document.documentElement.setAttribute('data-theme', t);
try{ localStorage.setItem(KEY, t); }catch(e){}
}
var saved = null;
try{ saved = localStorage.getItem(KEY); }catch(e){}
if (!saved) saved = window.matchMedia && window.matchMedia('(prefers-color-scheme: dark)').matches ? 'dark' : 'light';
apply(saved);
document.querySelector('.theme-toggle').addEventListener('click', function(){
apply(document.documentElement.getAttribute('data-theme') === 'dark' ? 'light' : 'dark');
});
// ===== back to top =====
var bt = document.getElementById('backTop');
window.addEventListener('scroll', function(){
if (window.scrollY > 400) bt.classList.add('show'); else bt.classList.remove('show');
});
bt.addEventListener('click', function(){ window.scrollTo({top:0, behavior:'smooth'}); });
})();
</script>
</body>
</html>