-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathconfig.py
More file actions
256 lines (223 loc) · 12.4 KB
/
Copy pathconfig.py
File metadata and controls
256 lines (223 loc) · 12.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
"""
实验配置文件
============
存储 API 密钥、模型列表、Prompt 模板、搜索模式定义。
修改模型或 Prompt 只需编辑此文件即可。
"""
import os
from dataclasses import dataclass, field
# ──────────────────────────────────────────────
# OpenRouter API 配置
# ──────────────────────────────────────────────
OPENROUTER_API_KEY: str = os.environ.get("OPENROUTER_API_KEY", "KEY")
OPENROUTER_BASE_URL: str = "https://openrouter.ai/api/v1/chat/completions"
# 请求头(OpenRouter 推荐附带 site 信息以便调试)
OPENROUTER_HEADERS: dict = {
"Authorization": f"Bearer {OPENROUTER_API_KEY}",
"Content-Type": "application/json",
"HTTP-Referer": "https://retracted-science-experiment.local",
"X-Title": "Retracted Science LLM Awareness Test",
}
# ──────────────────────────────────────────────
# 数据文件路径
# ──────────────────────────────────────────────
# 注意:这是 Spark 分区格式的 parquet 目录,pandas 可直接读取
DATA_PATH: str = "data/first_100_with_metadata.parquet"
# ──────────────────────────────────────────────
# 输出文件路径(JSONL 格式,实时追加)
# ──────────────────────────────────────────────
OUTPUT_DIR: str = "results"
# ──────────────────────────────────────────────
# 并发控制
# ──────────────────────────────────────────────
MAX_CONCURRENT_REQUESTS: int = 5 # 同时并发的最大请求数
REQUEST_TIMEOUT_SECONDS: float = 120.0 # 单次请求超时时间
# ──────────────────────────────────────────────
# 重试策略(指数退避)
# ──────────────────────────────────────────────
MAX_RETRIES: int = 5 # 最大重试次数
RETRY_BASE_DELAY: float = 2.0 # 基础等待秒数(实际 = base * 2^attempt)
RETRY_MAX_DELAY: float = 60.0 # 单次等待上限
# ──────────────────────────────────────────────
# 模型定义
# ──────────────────────────────────────────────
@dataclass
class ModelConfig:
"""单个模型的配置。"""
model_id: str # OpenRouter 模型 ID
display_name: str # 人类可读名称(用于日志/结果标识)
group: str # "open_source" 或 "closed_source"
supports_search: bool = False # 是否支持搜索模式(仅闭源模型)
max_tokens: int = 1024 # 最大生成 token 数(需足够容纳完整 JSON)
provider_preferences: dict = field(default_factory=dict) # 可选 provider 路由偏好
# --- 免费测试模型(用于验证 pipeline 是否跑通,不计入正式实验) ---
FREE_TEST_MODELS: list[ModelConfig] = [
ModelConfig(
model_id="google/gemma-3-27b-it:free",
display_name="Gemma-3-27B (free)",
group="free_test",
supports_search=True,
),
]
# --- 开源模型组 ---
OPEN_SOURCE_MODELS: list[ModelConfig] = [
ModelConfig(
model_id="openai/gpt-oss-20b",
display_name="GPT-OSS-20B",
group="open_source",
),
ModelConfig(
model_id="qwen/qwen3-32b",
display_name="Qwen3-32B",
group="open_source",
),
]
# --- 闭源模型组 ---
CLOSED_SOURCE_MODELS: list[ModelConfig] = [
ModelConfig(
model_id="openai/gpt-5.2",
display_name="GPT-5.2",
group="closed_source",
supports_search=True,
),
ModelConfig(
model_id="google/gemini-2.5-pro",
display_name="Gemini 2.5 Pro",
group="closed_source",
supports_search=True,
max_tokens=16384, # 需要更大预算以容纳强制 reasoning tokens
),
ModelConfig(
model_id="anthropic/claude-sonnet-4.6",
display_name="Claude Sonnet 4.6",
group="closed_source",
supports_search=True,
),
]
# 合并为全部模型列表(正式实验用)
ALL_MODELS: list[ModelConfig] = OPEN_SOURCE_MODELS + CLOSED_SOURCE_MODELS
# 仅免费模型(用于 --test-free 快速验证 pipeline)
TEST_MODELS: list[ModelConfig] = FREE_TEST_MODELS
# ──────────────────────────────────────────────
# Prompt 模板
# ──────────────────────────────────────────────
# {apa} → APA 引用文本
# {abstract} → 摘要文本(仅 Complex Prompt)
# {metadata} → DOI / 年份 / 引用数 / 主题(仅 Complex Prompt)
PROMPT_TEMPLATES: dict[str, str] = {
# ---- Simple Prompt: APA citation only ----
"simple": (
"Determine whether the following academic paper has been retracted.\n"
"APA citation: {apa}\n\n"
'Return ONLY the following JSON format, with no other text:\n'
'{{"is_retracted": "Yes"/"No"/"I don\'t know", "confidence": 0.0-1.0, "reasoning_summary": "brief reason"}}\n\n'
'The value of "is_retracted" MUST be exactly one of: "Yes", "No", or "I don\'t know".'
),
# ---- Complex Prompt: APA citation + abstract + metadata ----
"complex": (
"Based on the following information, determine whether this academic paper has been retracted.\n\n"
"APA citation: {apa}\n\n"
"Abstract:\n{abstract}\n\n"
"Additional metadata:\n{metadata}\n\n"
'Return ONLY the following JSON format, with no other text:\n'
'{{"is_retracted": "Yes"/"No"/"I don\'t know", "confidence": 0.0-1.0, "reasoning_summary": "brief reason"}}\n\n'
'The value of "is_retracted" MUST be exactly one of: "Yes", "No", or "I don\'t know".'
),
}
# ──────────────────────────────────────────────
# 搜索模式定义
# ──────────────────────────────────────────────
# search_mode 值: "none" | "auto" | "forced" | "hand-crafted"
#
# - none: 不使用任何搜索(开源模型默认且唯一模式)
# - auto: OpenRouter web search plugin(支持 native 的用 native,否则 fallback Exa),单次 API 调用
# - forced: 强制搜索 OpenAlex:Step 1 LLM 生成查询 → Step 2 调用 OpenAlex → Step 3 LLM 最终判断
# - hand-crafted: 从 APA 引用中提取 DOI(或 title)查 OpenAlex,将全部元数据提供给 LLM 判断
SEARCH_MODES_FOR_OPEN_SOURCE: list[str] = ["none"]
SEARCH_MODES_FOR_CLOSED_SOURCE: list[str] = ["none", "auto", "forced", "hand-crafted"]
# ──────────────────────────────────────────────
# Forced Search: 强制 OpenAlex 搜索流程
# ──────────────────────────────────────────────
# Step 1: 让 LLM 生成 OpenAlex 搜索查询(始终搜索,无跳过选项)
#
# {extra_context} 占位符:
# - simple prompt → 空字符串
# - complex prompt → "Abstract:\n{abstract}\n\nAdditional metadata:\n{metadata}\n"
FORCED_SEARCH_QUERY_PROMPT: str = (
"You need to search an academic database (OpenAlex) to determine whether the following paper has been retracted.\n\n"
"APA citation: {apa}\n"
"{extra_context}\n"
"Generate a search query to find this exact paper in the OpenAlex API.\n"
"The query will be used as a full-text search parameter, so use the paper title as the primary search terms.\n"
"You may also include author names or journal name if needed, but do NOT include DOIs, URLs, or API syntax.\n\n"
'Return ONLY the following JSON format, with no other text:\n'
'{{"search_query": "your search query here"}}'
)
# Step 3: 将 OpenAlex 搜索结果与原始 APA 引用一起提供给 LLM 做最终判断
FORCED_SEARCH_FINAL_PROMPT: str = (
"Determine whether the following academic paper has been retracted.\n"
"APA citation: {apa}\n\n"
"Below are search results from the OpenAlex academic database for this paper:\n"
"{search_results}\n\n"
'Return ONLY the following JSON format, with no other text:\n'
'{{"is_retracted": "Yes"/"No"/"I don\'t know", "confidence": 0.0-1.0, "reasoning_summary": "brief reason"}}\n\n'
'The value of "is_retracted" MUST be exactly one of: "Yes", "No", or "I don\'t know".'
)
# ──────────────────────────────────────────────
# Hand-Crafted: 从 APA 引用中提取 DOI/title 查 OpenAlex 全量元数据
# ──────────────────────────────────────────────
# {apa} → APA 引用文本
# {openalex_info} → OpenAlex 全量元数据格式化文本
HANDCRAFTED_PROMPT: str = (
"Determine whether the following academic paper has been retracted.\n"
"APA citation: {apa}\n\n"
"Below is the complete metadata retrieved from the OpenAlex academic database for this paper:\n"
"{openalex_info}\n\n"
'Return ONLY the following JSON format, with no other text:\n'
'{{"is_retracted": "Yes"/"No"/"I don\'t know", "confidence": 0.0-1.0, "reasoning_summary": "brief reason"}}\n\n'
'The value of "is_retracted" MUST be exactly one of: "Yes", "No", or "I don\'t know".'
)
# OpenAlex API 配置
OPENALEX_BASE_URL: str = "https://api.openalex.org/works"
OPENALEX_EMAIL: str = "yiti6755@colorado.edu"
OPENALEX_SELECT_FIELDS: str = "id,doi,display_name,publication_year,is_retracted,cited_by_count,abstract_inverted_index"
# ──────────────────────────────────────────────
# JSON 输出 Schema(强制模型遵循此结构)
# ──────────────────────────────────────────────
RESPONSE_JSON_SCHEMA: dict = {
"type": "object",
"properties": {
"is_retracted": {
"type": "string",
"enum": ["Yes", "No", "I don't know"],
"description": "Whether the paper is retracted: Yes, No, or I don't know.",
},
"confidence": {
"type": "number",
"description": "Your confidence from 0 to 1",
},
"reasoning_summary": {
"type": "string",
"description": "brief decision reason",
},
},
"required": ["is_retracted", "confidence", "reasoning_summary"],
}
# response_format 参数(传递给 OpenRouter)
RESPONSE_FORMAT: dict = {"type": "json_object"}
# ──────────────────────────────────────────────
# Reasoning / Thinking 控制
# ──────────────────────────────────────────────
# 禁用 thinking tokens 以节省成本并保持输出简洁
REASONING_CONFIG: dict = {
"exclude": True,
"max_tokens": 0,
}
# ──────────────────────────────────────────────
# Provider 偏好(全局默认,可被 ModelConfig 覆盖)
# ──────────────────────────────────────────────
DEFAULT_PROVIDER_PREFERENCES: dict = {
"allow_fallbacks": True,
"require_parameters": True,
"data_collection": "deny",
}