-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathreader.py
More file actions
559 lines (473 loc) · 23.4 KB
/
Copy pathreader.py
File metadata and controls
559 lines (473 loc) · 23.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
# -*- coding: UTF-8 -*-
"""
论文轻松阅读模块 - 核心阅读器
整合所有组件,提供完整的论文阅读功能。
"""
import os
import logging
import tempfile
from typing import List, Optional, Generator, AsyncGenerator
from .models import (
ReaderConfig,
PageInfo,
ImageRef,
CroppedImage,
StreamChunk,
ReadResult
)
from .pdf_converter import PDFConverter
from .image_processor import ImageProcessor
from .llm_client import PaperLLMClient
from .reference_parser import ReferenceParser
from .prompts import get_paper_reading_prompt, PRESET_PROMPTS
logger = logging.getLogger(__name__)
class PaperReader:
"""
论文轻松阅读器
将PDF论文转换为易于理解的内容,支持:
- PDF自动转图片
- 图片上传到云端
- 多模态LLM理解
- 图表智能引用
- 流式输出
使用示例:
```python
reader = PaperReader(api_key="your-api-key")
# 流式阅读
for chunk in reader.read_stream("paper.pdf"):
print(chunk.content, end="", flush=True)
# 同步阅读
result = reader.read("paper.pdf")
print(result.content)
```
"""
def __init__(
self,
api_key: str,
model: str = "glm-4.6v",
dpi: int = 150,
enable_thinking: bool = True,
**kwargs
):
"""
初始化论文阅读器
Args:
api_key: 智谱AI API密钥
model: 使用的模型名称
dpi: PDF转图片的DPI
enable_thinking: 是否启用思考模式
**kwargs: 其他配置参数
"""
self.config = ReaderConfig(
api_key=api_key,
model=model,
dpi=dpi,
enable_thinking=enable_thinking,
**kwargs
)
self.pdf_converter = PDFConverter(self.config)
self.image_processor = ImageProcessor(self.config)
self.llm_client = PaperLLMClient(self.config)
self.reference_parser = ReferenceParser()
# 缓存
self._pages: List[PageInfo] = []
self._output_dir: Optional[str] = None
def _prepare_pdf(
self,
pdf_path: str,
page_range: Optional[tuple] = None
) -> List[PageInfo]:
"""
准备PDF:转换为图片并上传
Args:
pdf_path: PDF文件路径
page_range: 页码范围
Returns:
PageInfo列表
"""
logger.info(f"准备PDF: {pdf_path}")
# 创建临时目录
self._output_dir = tempfile.mkdtemp(prefix="paper_reader_")
self._output_dir = 'tests/tmp2'
# 转换PDF为图片
logger.info("步骤1: 转换PDF为图片...")
pages = self.pdf_converter.convert(
pdf_path,
self._output_dir,
page_range
)
# 上传图片
logger.info("步骤2: 上传图片到云端...")
pages = self.image_processor.upload_pages(pages)
self._pages = pages
logger.info(f"PDF准备完成,共 {len(pages)} 页")
return pages
def read_stream(
self,
pdf_path: str,
prompt_style: str = "simple",
additional_instructions: str = "",
page_range: Optional[tuple] = None
) -> Generator[StreamChunk, None, None]:
"""
流式阅读论文
Args:
pdf_path: PDF文件路径
prompt_style: 提示词风格 (simple/detailed/summary/technical)
additional_instructions: 额外指令
page_range: 页码范围
Yields:
StreamChunk对象
"""
# 准备PDF
pages = self._prepare_pdf(pdf_path, page_range)
# 获取提示词
system_prompt, user_prompt = get_paper_reading_prompt(
prompt_style,
additional_instructions
)
# 获取图片URL
image_urls = [p.remote_url for p in pages if p.remote_url]
logger.info("步骤3: 调用LLM进行论文解读...")
# 流式调用LLM
yield from self.llm_client.chat_stream(
image_urls,
user_prompt,
system_prompt
)
def read(
self,
pdf_path: str,
prompt_style: str = "simple",
additional_instructions: str = "",
page_range: Optional[tuple] = None,
process_references: bool = True
) -> ReadResult:
"""
同步阅读论文
Args:
pdf_path: PDF文件路径
prompt_style: 提示词风格
additional_instructions: 额外指令
page_range: 页码范围
process_references: 是否处理图表引用
Returns:
ReadResult对象
"""
# 收集流式输出
thinking_parts: List[str] = []
answer_parts: List[str] = []
for chunk in self.read_stream(
pdf_path,
prompt_style,
additional_instructions,
page_range
):
if chunk.phase == "thinking":
thinking_parts.append(chunk.content)
elif chunk.phase == "answer":
answer_parts.append(chunk.content)
raw_content = "".join(answer_parts)
thinking = "".join(thinking_parts)
#
# thinking = """用户让我把这篇DeepSeek-V3.2的论文转换成轻松阅读模式,首先得抓住核心内容。首先,论文主要讲的是DeepSeek-V3.2这个大语言模型,它的三个关键技术:稀疏注意力(DSA)、可扩展的强化学习框架、大规模智能任务合成管道。然后要解释这些技术怎么让模型更高效、更聪明,还有它的性能表现。
#
# 首先,开头得用一两句话概括,比如“DeepSeek-V3.2是一个既高效又聪明的大语言模型,通过三个关键技术提升了推理能力和智能任务表现,甚至能和顶尖的闭源模型媲美。”然后分部分讲每个技术,用比喻,比如DSA像“给模型装了个智能过滤器,只关注重要的信息,减少计算负担”,强化学习框架像“给模型请了个教练,用更多计算让它更厉害”,合成管道像“给模型做了大量练习题,提升解决实际问题的能力”。
#
# 接下来,要引用图表,比如Figure 1是性能对比,得说明这个图展示了DeepSeek-V3.2在推理和智能任务上的表现,比其他模型好。Figure 2是DSA的架构,解释怎么选重要信息。Figure 3是推理成本,说明DSA怎么降低成本。Table 1是智能任务的描述,Table 2是和其他模型的对比,Table 3是推理模型的性能,Table 4是数学竞赛的成绩,Table 5是合成任务的效果,Figure 5是RL训练的效果,Figure 6是上下文管理的效果。
#
# 然后,结构要清晰,分几个部分:1. 核心贡献(三个技术);2. 每个技术的解释(用比喻);3. 性能表现(用图表支持);4. 总结。
#
# 要注意避免专业术语,比如“注意力机制”可以解释为“模型关注输入信息的方式”,“强化学习”是“通过奖励让模型学习”,“智能任务”是“用工具解决实际问题,比如搜索、编程”。
#
# 还要生动,比如DSA的闪电索引器像“快速扫描器,给每个信息打分,选分数高的”,token选择机制像“只保留最重要的信息,忽略无关的”。强化学习框架像“给模型更多计算资源,像给运动员加练,提升表现”。合成管道像“生成大量练习题,让模型在解决实际问题中进步”。
#
# 然后,性能部分,比如Figure 1显示DeepSeek-V3.2在AIME 2025、HMMT 2025等推理任务上得分很高,甚至超过GPT-5。Table 2显示在数学、代码、工具使用等任务上比其他开源模型好。Table 4显示在IMO、IOI等数学竞赛中拿到金牌,说明推理能力很强。
#
# 还要注意逻辑顺序,先讲模型解决了什么问题(开源模型效率低、计算不足、智能任务表现差),然后讲怎么解决的(三个技术),再讲效果(性能提升)。
#
# 最后总结,强调DeepSeek-V3.2是开源模型中的佼佼者,平衡了效率和性能,甚至能和闭源模型媲美,为开源社区提供了新的方向。
# # content..."""
# raw_content = """<|begin_of_box|># DeepSeek-V3.2:让大语言模型更高效、更聪明的“三件法宝”
#
#
# ## 核心内容一句话概括
# DeepSeek-V3.2是一个**既省计算资源又超会推理**的大语言模型,通过三个关键技术(稀疏注意力、可扩展强化学习、智能任务合成),解决了开源模型“效率低、算力不够、解决实际问题能力弱”的痛点,甚至能和顶尖闭源模型(如GPT-5、Gemini-3.0-Pro)掰手腕,还在数学竞赛中拿了金牌!
#
#
# ## 一、为什么需要DeepSeek-V3.2?
# 开源大语言模型(比如我们常见的免费模型)有个“三宗罪”:
# 1. **效率低**:处理长文本时,像“逐字读小说”一样慢,计算资源浪费严重;
# 2. **算力不够**:训练时“吃”的计算资源少,导致解决复杂问题(如数学、编程)的能力弱;
# 3. **智能任务差**:用工具(比如搜索、编程)解决实际问题时,经常“卡壳”,不如闭源模型(如GPT-4)好用。
#
# DeepSeek-V3.2就是来解决这些问题的“救星”!
#
#
# ## 二、DeepSeek-V3.2的“三件法宝”
# ### 1. 稀疏注意力(DSA):给模型装个“智能过滤器”
# 大语言模型处理文本时,需要“关注”输入中的所有信息,但很多信息其实不重要(比如“的、地、得”这类虚词)。DSA就像给模型装了个**快速扫描器**:
# - 先用“闪电索引器”给每个信息打分(比如“关键名词”分数高,“虚词”分数低);
# - 再用“token选择机制”只保留分数最高的信息,忽略无关内容。
#
# 这样既减少了计算负担,又不影响模型理解文本!
# **效果**:处理长文本时,计算成本从“平方级”(比如1000个词要算100万次)降到“线性级”(1000个词只算1000次),速度提升好几倍!
# 看!!<image_ref>[4, [[142, 93, 856, 343]], "Figure 2: Attention architecture of DeepSeek-V3.2, where DSA is instantiated under MLA. The green part illustrates how DSA selects the top-k key-value entries according to the indexer."]</image_ref> 展示了DSA怎么选重要信息,绿色部分就是“过滤器”在工作。
#
#
# ### 2. 可扩展强化学习框架:给模型请个“金牌教练”
# 强化学习就像“给模型奖励”,让它学会做对的事(比如推理正确、用工具正确)。但开源模型往往“训练不够”,因为算力有限。DeepSeek-V3.2的框架像**给模型请了个金牌教练**:
# - 用更稳定的训练方法(比如“无偏KL估计”“离策略序列掩码”),避免训练时“走弯路”;
# - 增加“后训练”的计算资源(超过预训练的10%),让模型“加练”更多任务。
#
# **效果**:模型推理能力大幅提升,甚至能和GPT-5媲美!比如在数学推理任务(AIME 2025)中,得分96.0%,超过GPT-5的94.6%(见!<image_ref>[1, [[117, 546, 877, 837]], "Figure 1: Benchmark of DeepSeek-V3.2 and its counterparts. For HMMT 2025, we report the February competition, consistent with the baselines. For HLE, we report the text-only subset."]</image_ref>)。
#
#
# ### 3. 大规模智能任务合成管道:给模型做“海量练习题”
# 模型解决实际问题时(比如用搜索工具找信息、用编程工具写代码),需要“实战经验”。DeepSeek-V3.2的合成管道像**给模型做了8.5万道练习题**:
# - 生成1800多个不同的“环境”(比如模拟搜索、编程场景);
# - 让模型在这些环境中“练习”用工具,提升“举一反三”的能力。
#
# **效果**:模型在智能任务(如工具使用、代码解释)上的表现大幅提升,比如在“工具十项全能”(Tool-Decathlon)任务中,得分35.2%,比其他开源模型高很多(见!<table_ref>[10, [[223, 502, 774, 586]], "Table 1: The description of different agent tasks, including the number of tasks, environment type (real or synthesized), and prompt source (extracted or synthesized)."]</table_ref>)。
#
#
# ## 三、DeepSeek-V3.2到底有多强?
# ### 1. 推理能力:数学竞赛拿金牌!
# 在2025年国际数学奥林匹克(IMO)、信息学奥林匹克(IOI)等顶级竞赛中,DeepSeek-V3.2-Speciale(高算力版本)拿了**金牌**!比如在IMO 2025中,解决了35/42道题,超过Gemini-3.0-Pro(见!<table_ref>[15, [[223, 452, 776, 592]], "Table 4: Performance of DeepSeek-V3.2-Speciale in top-tier mathematics and coding competitions. For ICPC WF 2025, we report the number of submissions for each successfully solved problem. DeepSeek-V3.2-Speciale ranked 2nd in ICPC WF 2025 and 10th in IOI 2025."]</table_ref>)。
#
#
# ### 2. 智能任务:用工具更顺手
# 在“搜索代理”“代码代理”等任务中,DeepSeek-V3.2的表现远超其他开源模型。比如在“搜索代理”任务(用搜索工具找信息)中,得分51.4%,比MiniMax M2(44.0%)高很多(见!<table_ref>[13, [[127, 421, 875, 715]], "Table 2: Comparison between DeepSeek-V3.2 and closed/open models. For open models, we just compare with models supports thinking in tooluse. Numbers in bold represent the best scores within each model class (open-source and closed-source). The τ²-Bench result is computed by the average of each category. Regarding BrowseComp, the performance with the context management technique is noted with *."]</table_ref>)。
#
#
# ### 3. 效率:省计算资源,还快!
# DSA让模型处理长文本时更省资源。比如在“预填充”(处理输入文本)时,DeepSeek-V3.2的成本比旧版本(DeepSeek-V3.1-Terminus)低很多(见!<image_ref>[6, [[119, 102, 860, 317]], "Figure 3: Inference costs of DeepSeek-V3.1-Terminus and DeepSeek-V3.2 on H800 clusters."]</image_ref>),处理128K长文本时,成本只有旧版本的1/3!
#
#
# ## 四、总结:开源模型的“新标杆”
# DeepSeek-V3.2通过“稀疏注意力(省资源)+ 强化学习(提能力)+ 合成任务(练实战)”,解决了开源模型的“效率-性能”矛盾,成为开源社区中的“佼佼者”。它的出现,让开源模型也能和闭源模型“平起平坐”,甚至在一些任务上超过闭源模型,为AI的发展提供了新的方向!
#
#
# **关键图表回顾**:
# - !<image_ref>[1, [[117, 546, 877, 837]], "Figure 1: Benchmark of DeepSeek-V3.2 and its counterparts. For HMMT 2025, we report the February competition, consistent with the baselines. For HLE, we report the text-only subset."]</image_ref>:展示DeepSeek-V3.2在推理和智能任务上的性能,超过GPT-5等模型;
# - !<image_ref>[4, [[143, 88, 861, 366]], "Figure 2: Attention architecture of DeepSeek-V3.2, where DSA is instantiated under MLA. The green part illustrates how DSA selects the top-k key-value entries according to the indexer."]</image_ref>:解释DSA如何选重要信息,提升效率;
# - !<image_ref>[6, [[119, 102, 860, 317]], "Figure 3: Inference costs of DeepSeek-V3.1-Terminus and DeepSeek-V3.2 on H800 clusters."]</image_ref>:展示DSA降低推理成本的效果;
# - !<table_ref>[15, [[223, 452, 776, 592]], "Table 4: Performance of DeepSeek-V3.2-Speciale in top-tier mathematics and coding competitions. For ICPC WF 2025, we report the number of submissions for each successfully solved problem. DeepSeek-V3.2-Speciale ranked 2nd in ICPC WF 2025 and 10th in IOI 2025."]</table_ref>:展示DeepSeek-V3.2在数学竞赛中的金牌成绩。<|end_of_box|>
# """
result = ReadResult(
content=raw_content,
thinking=thinking
)
print('# thinking...')
print(result.thinking)
print('# content...')
print(result.content)
# 处理图表引用
if process_references:
result = self._process_references(result)
return result
def _process_references(self, result: ReadResult) -> ReadResult:
"""
处理结果中的图表引用(ReAct Round 2)
这是 ReAct 模式的第二阶段:
1. 解析第一轮输出中的图表引用标记
2. 根据坐标裁剪图片并上传
3. 调用 LLM 第二轮,整合图片 URL 生成最终内容
Args:
result: 原始阅读结果(第一轮输出)
Returns:
处理后的结果(最终输出)
"""
logger.info("步骤4: 处理图表引用 (ReAct Round 2)...")
# 解析引用
refs = self.reference_parser.parse_references(result.content)
result.image_refs = refs
if not refs:
logger.info("未找到图表引用,跳过第二轮处理")
return result
# 裁剪并上传引用的图片
logger.info(f"裁剪并上传 {len(refs)} 个图表...")
cropped_images = self.image_processor.process_references(
self._pages,
refs
)
result.cropped_images = cropped_images
if not cropped_images:
logger.warning("没有成功处理的图表,跳过第二轮 LLM 调用")
return result
# 第二轮 LLM 调用:整合图片 URL,生成最终内容
logger.info("步骤5: 第二轮 LLM 调用,整合图表...")
result.content = self.llm_client.chat_round2(
first_round_content=result.content,
cropped_images=cropped_images
)
logger.info(f"处理完成,共处理 {len(cropped_images)} 个引用")
return result
async def aread_stream(
self,
pdf_path: str,
prompt_style: str = "simple",
additional_instructions: str = "",
page_range: Optional[tuple] = None
) -> AsyncGenerator[StreamChunk, None]:
"""
异步流式阅读论文
Args:
pdf_path: PDF文件路径
prompt_style: 提示词风格
additional_instructions: 额外指令
page_range: 页码范围
Yields:
StreamChunk对象
"""
import asyncio
# 准备PDF(在线程池中执行)
loop = asyncio.get_event_loop()
pages = await loop.run_in_executor(
None,
lambda: self._prepare_pdf(pdf_path, page_range)
)
# 获取提示词
system_prompt, user_prompt = get_paper_reading_prompt(
prompt_style,
additional_instructions
)
# 获取图片URL
image_urls = [p.remote_url for p in pages if p.remote_url]
logger.info("步骤3: 调用LLM进行论文解读...")
# 异步流式调用
async for chunk in self.llm_client.achat_stream(
image_urls,
user_prompt,
system_prompt
):
yield chunk
async def aread(
self,
pdf_path: str,
prompt_style: str = "simple",
additional_instructions: str = "",
page_range: Optional[tuple] = None,
process_references: bool = True
) -> ReadResult:
"""
异步阅读论文
Args:
pdf_path: PDF文件路径
prompt_style: 提示词风格
additional_instructions: 额外指令
page_range: 页码范围
process_references: 是否处理图表引用
Returns:
ReadResult对象
"""
import asyncio
# 收集流式输出
thinking_parts: List[str] = []
answer_parts: List[str] = []
async for chunk in self.aread_stream(
pdf_path,
prompt_style,
additional_instructions,
page_range
):
if chunk.phase == "thinking":
thinking_parts.append(chunk.content)
elif chunk.phase == "answer":
answer_parts.append(chunk.content)
raw_content = "".join(answer_parts)
thinking = "".join(thinking_parts)
result = ReadResult(
content=raw_content,
thinking=thinking
)
# 处理图表引用
if process_references:
loop = asyncio.get_event_loop()
result = await loop.run_in_executor(
None,
lambda: self._process_references(result)
)
return result
def read_with_preset(
self,
pdf_path: str,
preset: str = "easy_read",
page_range: Optional[tuple] = None
) -> ReadResult:
"""
使用预设风格阅读论文
Args:
pdf_path: PDF文件路径
preset: 预设名称 (easy_read/quick_summary/deep_dive/find_figures)
page_range: 页码范围
Returns:
ReadResult对象
"""
if preset not in PRESET_PROMPTS:
raise ValueError(f"未知的预设: {preset},可用: {list(PRESET_PROMPTS.keys())}")
preset_config = PRESET_PROMPTS[preset]
# 准备PDF
pages = self._prepare_pdf(pdf_path, page_range)
# 获取图片URL
image_urls = [p.remote_url for p in pages if p.remote_url]
# 调用LLM
raw_content = self.llm_client.chat(
image_urls,
preset_config["user"],
preset_config["system"]
)
result = ReadResult(content=raw_content)
# 处理引用
return self._process_references(result)
def cleanup(self):
"""清理临时文件"""
import shutil
if self._output_dir and os.path.exists(self._output_dir):
try:
shutil.rmtree(self._output_dir)
logger.info(f"已清理临时目录: {self._output_dir}")
except Exception as e:
logger.warning(f"清理临时目录失败: {e}")
self._output_dir = None
self._pages = []
def __enter__(self):
return self
def __exit__(self, exc_type, exc_val, exc_tb):
self.cleanup()
return False
def read_paper(
pdf_path: str,
api_key: str,
prompt_style: str = "simple"
) -> str:
"""
便捷函数:阅读论文
Args:
pdf_path: PDF文件路径
api_key: 智谱AI API密钥
prompt_style: 提示词风格
Returns:
解读后的内容
"""
with PaperReader(api_key=api_key) as reader:
result = reader.read(pdf_path, prompt_style)
return result.content
def read_paper_stream(
pdf_path: str,
api_key: str,
prompt_style: str = "simple"
) -> Generator[str, None, None]:
"""
便捷函数:流式阅读论文
Args:
pdf_path: PDF文件路径
api_key: 智谱AI API密钥
prompt_style: 提示词风格
Yields:
内容片段
"""
with PaperReader(api_key=api_key) as reader:
for chunk in reader.read_stream(pdf_path, prompt_style):
if chunk.phase in ["thinking", "answer"]:
yield chunk.content